From 77d8c0d4b32605de78e75f68b7fe886a5b1553d8 Mon Sep 17 00:00:00 2001 From: Michael Hansen Date: Sat, 3 Feb 2024 11:05:58 -0600 Subject: [PATCH 01/16] Add configs for mls models --- models/de_DE-mls-medium.pt.json | 740 ++++++++++++++++++++++++++++++++ models/fr_FR-mls-medium.pt.json | 629 +++++++++++++++++++++++++++ models/nl_NL-mls-medium.pt.json | 554 ++++++++++++++++++++++++ 3 files changed, 1923 insertions(+) create mode 100755 models/de_DE-mls-medium.pt.json create mode 100755 models/fr_FR-mls-medium.pt.json create mode 100755 models/nl_NL-mls-medium.pt.json diff --git a/models/de_DE-mls-medium.pt.json b/models/de_DE-mls-medium.pt.json new file mode 100755 index 0000000..0f50a27 --- /dev/null +++ b/models/de_DE-mls-medium.pt.json @@ -0,0 +1,740 @@ +{ + "dataset": "mls", + "audio": { + "sample_rate": 22050, + "quality": "medium" + }, + "espeak": { + "voice": "de" + }, + "language": { + "code": "de_DE" + }, + "inference": { + "noise_scale": 0.333, + "length_scale": 1, + "noise_w": 0.333 + }, + "phoneme_type": "espeak", + "phoneme_map": {}, + "phoneme_id_map": { + " ": [ + 3 + ], + "!": [ + 4 + ], + "\"": [ + 150 + ], + "#": [ + 149 + ], + "$": [ + 2 + ], + "'": [ + 5 + ], + "(": [ + 6 + ], + ")": [ + 7 + ], + ",": [ + 8 + ], + "-": [ + 9 + ], + ".": [ + 10 + ], + "0": [ + 130 + ], + "1": [ + 131 + ], + "2": [ + 132 + ], + "3": [ + 133 + ], + "4": [ + 134 + ], + "5": [ + 135 + ], + "6": [ + 136 + ], + "7": [ + 137 + ], + "8": [ + 138 + ], + "9": [ + 139 + ], + ":": [ + 11 + ], + ";": [ + 12 + ], + "?": [ + 13 + ], + "X": [ + 156 + ], + "^": [ + 1 + ], + "_": [ + 0 + ], + "a": [ + 14 + ], + "b": [ + 15 + ], + "c": [ + 16 + ], + "d": [ + 17 + ], + "e": [ + 18 + ], + "f": [ + 19 + ], + "g": [ + 154 + ], + "h": [ + 20 + ], + "i": [ + 21 + ], + "j": [ + 22 + ], + "k": [ + 23 + ], + "l": [ + 24 + ], + "m": [ + 25 + ], + "n": [ + 26 + ], + "o": [ + 27 + ], + "p": [ + 28 + ], + "q": [ + 29 + ], + "r": [ + 30 + ], + "s": [ + 31 + ], + "t": [ + 32 + ], + "u": [ + 33 + ], + "v": [ + 34 + ], + "w": [ + 35 + ], + "x": [ + 36 + ], + "y": [ + 37 + ], + "z": [ + 38 + ], + "æ": [ + 39 + ], + "ç": [ + 40 + ], + "ð": [ + 41 + ], + "ø": [ + 42 + ], + "ħ": [ + 43 + ], + "ŋ": [ + 44 + ], + "œ": [ + 45 + ], + "ǀ": [ + 46 + ], + "ǁ": [ + 47 + ], + "ǂ": [ + 48 + ], + "ǃ": [ + 49 + ], + "ɐ": [ + 50 + ], + "ɑ": [ + 51 + ], + "ɒ": [ + 52 + ], + "ɓ": [ + 53 + ], + "ɔ": [ + 54 + ], + "ɕ": [ + 55 + ], + "ɖ": [ + 56 + ], + "ɗ": [ + 57 + ], + "ɘ": [ + 58 + ], + "ə": [ + 59 + ], + "ɚ": [ + 60 + ], + "ɛ": [ + 61 + ], + "ɜ": [ + 62 + ], + "ɞ": [ + 63 + ], + "ɟ": [ + 64 + ], + "ɠ": [ + 65 + ], + "ɡ": [ + 66 + ], + "ɢ": [ + 67 + ], + "ɣ": [ + 68 + ], + "ɤ": [ + 69 + ], + "ɥ": [ + 70 + ], + "ɦ": [ + 71 + ], + "ɧ": [ + 72 + ], + "ɨ": [ + 73 + ], + "ɪ": [ + 74 + ], + "ɫ": [ + 75 + ], + "ɬ": [ + 76 + ], + "ɭ": [ + 77 + ], + "ɮ": [ + 78 + ], + "ɯ": [ + 79 + ], + "ɰ": [ + 80 + ], + "ɱ": [ + 81 + ], + "ɲ": [ + 82 + ], + "ɳ": [ + 83 + ], + "ɴ": [ + 84 + ], + "ɵ": [ + 85 + ], + "ɶ": [ + 86 + ], + "ɸ": [ + 87 + ], + "ɹ": [ + 88 + ], + "ɺ": [ + 89 + ], + "ɻ": [ + 90 + ], + "ɽ": [ + 91 + ], + "ɾ": [ + 92 + ], + "ʀ": [ + 93 + ], + "ʁ": [ + 94 + ], + "ʂ": [ + 95 + ], + "ʃ": [ + 96 + ], + "ʄ": [ + 97 + ], + "ʈ": [ + 98 + ], + "ʉ": [ + 99 + ], + "ʊ": [ + 100 + ], + "ʋ": [ + 101 + ], + "ʌ": [ + 102 + ], + "ʍ": [ + 103 + ], + "ʎ": [ + 104 + ], + "ʏ": [ + 105 + ], + "ʐ": [ + 106 + ], + "ʑ": [ + 107 + ], + "ʒ": [ + 108 + ], + "ʔ": [ + 109 + ], + "ʕ": [ + 110 + ], + "ʘ": [ + 111 + ], + "ʙ": [ + 112 + ], + "ʛ": [ + 113 + ], + "ʜ": [ + 114 + ], + "ʝ": [ + 115 + ], + "ʟ": [ + 116 + ], + "ʡ": [ + 117 + ], + "ʢ": [ + 118 + ], + "ʦ": [ + 155 + ], + "ʰ": [ + 145 + ], + "ʲ": [ + 119 + ], + "ˈ": [ + 120 + ], + "ˌ": [ + 121 + ], + "ː": [ + 122 + ], + "ˑ": [ + 123 + ], + "˞": [ + 124 + ], + "ˤ": [ + 146 + ], + "̃": [ + 141 + ], + "̊": [ + 158 + ], + "̝": [ + 157 + ], + "̧": [ + 140 + ], + "̩": [ + 144 + ], + "̪": [ + 142 + ], + "̯": [ + 143 + ], + "̺": [ + 152 + ], + "̻": [ + 153 + ], + "β": [ + 125 + ], + "ε": [ + 147 + ], + "θ": [ + 126 + ], + "χ": [ + 127 + ], + "ᵻ": [ + 128 + ], + "↑": [ + 151 + ], + "↓": [ + 148 + ], + "ⱱ": [ + 129 + ] + }, + "num_symbols": 256, + "num_speakers": 236, + "speaker_id_map": { + "2422": 0, + "4536": 1, + "2037": 2, + "9565": 3, + "10148": 4, + "6507": 5, + "5055": 6, + "3503": 7, + "252": 8, + "9132": 9, + "3990": 10, + "5753": 11, + "5424": 12, + "2602": 13, + "4174": 14, + "3885": 15, + "12415": 16, + "8470": 17, + "11927": 18, + "9639": 19, + "3494": 20, + "2946": 21, + "5283": 22, + "4533": 23, + "2497": 24, + "12275": 25, + "1649": 26, + "146": 27, + "8337": 28, + "4542": 29, + "589": 30, + "1998": 31, + "3797": 32, + "5244": 33, + "7328": 34, + "7998": 35, + "10179": 36, + "9610": 37, + "20": 38, + "253": 39, + "12899": 40, + "7194": 41, + "3759": 42, + "2677": 43, + "6719": 44, + "1897": 45, + "11990": 46, + "6880": 47, + "19": 48, + "9515": 49, + "327": 50, + "3244": 51, + "5324": 52, + "2234": 53, + "3124": 54, + "2043": 55, + "143": 56, + "8139": 57, + "9646": 58, + "8659": 59, + "9538": 60, + "989": 61, + "5405": 62, + "10087": 63, + "8294": 64, + "4396": 65, + "1474": 66, + "139": 67, + "136": 68, + "10791": 69, + "7242": 70, + "3631": 71, + "9908": 72, + "7906": 73, + "1171": 74, + "7479": 75, + "5632": 76, + "3731": 77, + "4650": 78, + "135": 79, + "145": 80, + "137": 81, + "1757": 82, + "91": 83, + "9514": 84, + "13494": 85, + "1946": 86, + "3277": 87, + "5595": 88, + "278": 89, + "7120": 90, + "7406": 91, + "11695": 92, + "1593": 93, + "3862": 94, + "138": 95, + "141": 96, + "9948": 97, + "1163": 98, + "1054": 99, + "1844": 100, + "4911": 101, + "7261": 102, + "8223": 103, + "7624": 104, + "144": 105, + "13871": 106, + "2974": 107, + "5934": 108, + "7002": 109, + "8769": 110, + "3363": 111, + "3040": 112, + "6067": 113, + "9494": 114, + "8743": 115, + "13255": 116, + "1660": 117, + "3588": 118, + "4748": 119, + "8450": 120, + "5295": 121, + "4705": 122, + "8125": 123, + "7272": 124, + "7320": 125, + "1874": 126, + "1262": 127, + "10870": 128, + "12379": 129, + "4463": 130, + "10349": 131, + "2252": 132, + "8325": 133, + "3052": 134, + "4001": 135, + "7456": 136, + "140": 137, + "4739": 138, + "11299": 139, + "4730": 140, + "6659": 141, + "2034": 142, + "2732": 143, + "2158": 144, + "3698": 145, + "5675": 146, + "6315": 147, + "10904": 148, + "7202": 149, + "11480": 150, + "10625": 151, + "11546": 152, + "4576": 153, + "4512": 154, + "8634": 155, + "13626": 156, + "1613": 157, + "287": 158, + "9207": 159, + "6982": 160, + "1724": 161, + "10191": 162, + "1091": 163, + "2909": 164, + "1965": 165, + "2506": 166, + "4414": 167, + "5764": 168, + "12776": 169, + "1033": 170, + "13726": 171, + "2314": 172, + "6826": 173, + "9706": 174, + "8427": 175, + "9168": 176, + "9287": 177, + "6905": 178, + "4153": 179, + "3330": 180, + "2859": 181, + "5406": 182, + "2840": 183, + "1920": 184, + "9241": 185, + "10163": 186, + "8305": 187, + "12461": 188, + "3276": 189, + "11413": 190, + "10536": 191, + "10614": 192, + "7579": 193, + "8675": 194, + "7483": 195, + "7270": 196, + "8704": 197, + "4468": 198, + "6611": 199, + "11497": 200, + "11772": 201, + "2792": 202, + "11481": 203, + "10162": 204, + "10819": 205, + "8732": 206, + "11328": 207, + "11920": 208, + "6646": 209, + "7486": 210, + "11870": 211, + "12417": 212, + "10364": 213, + "6117": 214, + "6448": 215, + "10433": 216, + "7515": 217, + "5823": 218, + "8567": 219, + "10947": 220, + "11869": 221, + "12335": 222, + "12500": 223, + "13755": 224, + "7006": 225, + "3685": 226, + "5819": 227, + "9353": 228, + "11355": 229, + "12174": 230, + "7150": 231, + "6952": 232, + "11987": 233, + "3995": 234, + "7449": 235 + }, + "piper_version": "1.0.0" +} diff --git a/models/fr_FR-mls-medium.pt.json b/models/fr_FR-mls-medium.pt.json new file mode 100755 index 0000000..ab1bb41 --- /dev/null +++ b/models/fr_FR-mls-medium.pt.json @@ -0,0 +1,629 @@ +{ + "dataset": "mls", + "audio": { + "sample_rate": 22050, + "quality": "medium" + }, + "espeak": { + "voice": "fr" + }, + "language": { + "code": "fr_FR" + }, + "inference": { + "noise_scale": 0.333, + "length_scale": 1, + "noise_w": 0.333 + }, + "phoneme_type": "espeak", + "phoneme_map": {}, + "phoneme_id_map": { + " ": [ + 3 + ], + "!": [ + 4 + ], + "\"": [ + 150 + ], + "#": [ + 149 + ], + "$": [ + 2 + ], + "'": [ + 5 + ], + "(": [ + 6 + ], + ")": [ + 7 + ], + ",": [ + 8 + ], + "-": [ + 9 + ], + ".": [ + 10 + ], + "0": [ + 130 + ], + "1": [ + 131 + ], + "2": [ + 132 + ], + "3": [ + 133 + ], + "4": [ + 134 + ], + "5": [ + 135 + ], + "6": [ + 136 + ], + "7": [ + 137 + ], + "8": [ + 138 + ], + "9": [ + 139 + ], + ":": [ + 11 + ], + ";": [ + 12 + ], + "?": [ + 13 + ], + "X": [ + 156 + ], + "^": [ + 1 + ], + "_": [ + 0 + ], + "a": [ + 14 + ], + "b": [ + 15 + ], + "c": [ + 16 + ], + "d": [ + 17 + ], + "e": [ + 18 + ], + "f": [ + 19 + ], + "g": [ + 154 + ], + "h": [ + 20 + ], + "i": [ + 21 + ], + "j": [ + 22 + ], + "k": [ + 23 + ], + "l": [ + 24 + ], + "m": [ + 25 + ], + "n": [ + 26 + ], + "o": [ + 27 + ], + "p": [ + 28 + ], + "q": [ + 29 + ], + "r": [ + 30 + ], + "s": [ + 31 + ], + "t": [ + 32 + ], + "u": [ + 33 + ], + "v": [ + 34 + ], + "w": [ + 35 + ], + "x": [ + 36 + ], + "y": [ + 37 + ], + "z": [ + 38 + ], + "æ": [ + 39 + ], + "ç": [ + 40 + ], + "ð": [ + 41 + ], + "ø": [ + 42 + ], + "ħ": [ + 43 + ], + "ŋ": [ + 44 + ], + "œ": [ + 45 + ], + "ǀ": [ + 46 + ], + "ǁ": [ + 47 + ], + "ǂ": [ + 48 + ], + "ǃ": [ + 49 + ], + "ɐ": [ + 50 + ], + "ɑ": [ + 51 + ], + "ɒ": [ + 52 + ], + "ɓ": [ + 53 + ], + "ɔ": [ + 54 + ], + "ɕ": [ + 55 + ], + "ɖ": [ + 56 + ], + "ɗ": [ + 57 + ], + "ɘ": [ + 58 + ], + "ə": [ + 59 + ], + "ɚ": [ + 60 + ], + "ɛ": [ + 61 + ], + "ɜ": [ + 62 + ], + "ɞ": [ + 63 + ], + "ɟ": [ + 64 + ], + "ɠ": [ + 65 + ], + "ɡ": [ + 66 + ], + "ɢ": [ + 67 + ], + "ɣ": [ + 68 + ], + "ɤ": [ + 69 + ], + "ɥ": [ + 70 + ], + "ɦ": [ + 71 + ], + "ɧ": [ + 72 + ], + "ɨ": [ + 73 + ], + "ɪ": [ + 74 + ], + "ɫ": [ + 75 + ], + "ɬ": [ + 76 + ], + "ɭ": [ + 77 + ], + "ɮ": [ + 78 + ], + "ɯ": [ + 79 + ], + "ɰ": [ + 80 + ], + "ɱ": [ + 81 + ], + "ɲ": [ + 82 + ], + "ɳ": [ + 83 + ], + "ɴ": [ + 84 + ], + "ɵ": [ + 85 + ], + "ɶ": [ + 86 + ], + "ɸ": [ + 87 + ], + "ɹ": [ + 88 + ], + "ɺ": [ + 89 + ], + "ɻ": [ + 90 + ], + "ɽ": [ + 91 + ], + "ɾ": [ + 92 + ], + "ʀ": [ + 93 + ], + "ʁ": [ + 94 + ], + "ʂ": [ + 95 + ], + "ʃ": [ + 96 + ], + "ʄ": [ + 97 + ], + "ʈ": [ + 98 + ], + "ʉ": [ + 99 + ], + "ʊ": [ + 100 + ], + "ʋ": [ + 101 + ], + "ʌ": [ + 102 + ], + "ʍ": [ + 103 + ], + "ʎ": [ + 104 + ], + "ʏ": [ + 105 + ], + "ʐ": [ + 106 + ], + "ʑ": [ + 107 + ], + "ʒ": [ + 108 + ], + "ʔ": [ + 109 + ], + "ʕ": [ + 110 + ], + "ʘ": [ + 111 + ], + "ʙ": [ + 112 + ], + "ʛ": [ + 113 + ], + "ʜ": [ + 114 + ], + "ʝ": [ + 115 + ], + "ʟ": [ + 116 + ], + "ʡ": [ + 117 + ], + "ʢ": [ + 118 + ], + "ʦ": [ + 155 + ], + "ʰ": [ + 145 + ], + "ʲ": [ + 119 + ], + "ˈ": [ + 120 + ], + "ˌ": [ + 121 + ], + "ː": [ + 122 + ], + "ˑ": [ + 123 + ], + "˞": [ + 124 + ], + "ˤ": [ + 146 + ], + "̃": [ + 141 + ], + "̊": [ + 158 + ], + "̝": [ + 157 + ], + "̧": [ + 140 + ], + "̩": [ + 144 + ], + "̪": [ + 142 + ], + "̯": [ + 143 + ], + "̺": [ + 152 + ], + "̻": [ + 153 + ], + "β": [ + 125 + ], + "ε": [ + 147 + ], + "θ": [ + 126 + ], + "χ": [ + 127 + ], + "ᵻ": [ + 128 + ], + "↑": [ + 151 + ], + "↓": [ + 148 + ], + "ⱱ": [ + 129 + ] + }, + "num_symbols": 256, + "num_speakers": 125, + "speaker_id_map": { + "1840": 0, + "3698": 1, + "123": 2, + "1474": 3, + "12709": 4, + "7423": 5, + "9242": 6, + "8778": 7, + "3060": 8, + "4512": 9, + "6249": 10, + "12541": 11, + "13634": 12, + "10065": 13, + "6128": 14, + "5232": 15, + "5764": 16, + "12713": 17, + "12823": 18, + "6070": 19, + "12501": 20, + "9121": 21, + "1649": 22, + "2776": 23, + "11772": 24, + "5612": 25, + "11822": 26, + "1590": 27, + "5525": 28, + "10827": 29, + "1243": 30, + "13142": 31, + "62": 32, + "13177": 33, + "10620": 34, + "8102": 35, + "8582": 36, + "11875": 37, + "7239": 38, + "9854": 39, + "7377": 40, + "10082": 41, + "12512": 42, + "1329": 43, + "2506": 44, + "6856": 45, + "10058": 46, + "103": 47, + "14": 48, + "6381": 49, + "1664": 50, + "11954": 51, + "66": 52, + "1127": 53, + "3270": 54, + "13611": 55, + "13658": 56, + "12968": 57, + "1989": 58, + "12981": 59, + "7193": 60, + "6348": 61, + "7679": 62, + "2284": 63, + "3182": 64, + "3503": 65, + "2033": 66, + "2771": 67, + "7614": 68, + "125": 69, + "3204": 70, + "5595": 71, + "5553": 72, + "694": 73, + "1624": 74, + "1887": 75, + "2926": 76, + "7150": 77, + "3190": 78, + "3344": 79, + "4699": 80, + "1798": 81, + "1745": 82, + "5077": 83, + "753": 84, + "52": 85, + "4174": 86, + "4018": 87, + "12899": 88, + "1844": 89, + "4396": 90, + "1817": 91, + "2155": 92, + "2946": 93, + "4336": 94, + "4609": 95, + "1977": 96, + "10957": 97, + "204": 98, + "4650": 99, + "5295": 100, + "5968": 101, + "4744": 102, + "2825": 103, + "9804": 104, + "707": 105, + "30": 106, + "115": 107, + "5840": 108, + "2587": 109, + "2607": 110, + "2544": 111, + "28": 112, + "27": 113, + "177": 114, + "112": 115, + "94": 116, + "2596": 117, + "3595": 118, + "7032": 119, + "7848": 120, + "11247": 121, + "7439": 122, + "2904": 123, + "6362": 124 + }, + "piper_version": "1.0.0" +} diff --git a/models/nl_NL-mls-medium.pt.json b/models/nl_NL-mls-medium.pt.json new file mode 100755 index 0000000..5673d00 --- /dev/null +++ b/models/nl_NL-mls-medium.pt.json @@ -0,0 +1,554 @@ +{ + "audio": { + "sample_rate": 22050 + }, + "espeak": { + "voice": "nl" + }, + "language": { + "code": "nl_NL" + }, + "inference": { + "noise_scale": 0.333, + "length_scale": 1, + "noise_w": 0.333 + }, + "phoneme_type": "espeak", + "phoneme_map": {}, + "phoneme_id_map": { + " ": [ + 3 + ], + "!": [ + 4 + ], + "\"": [ + 150 + ], + "#": [ + 149 + ], + "$": [ + 2 + ], + "'": [ + 5 + ], + "(": [ + 6 + ], + ")": [ + 7 + ], + ",": [ + 8 + ], + "-": [ + 9 + ], + ".": [ + 10 + ], + "0": [ + 130 + ], + "1": [ + 131 + ], + "2": [ + 132 + ], + "3": [ + 133 + ], + "4": [ + 134 + ], + "5": [ + 135 + ], + "6": [ + 136 + ], + "7": [ + 137 + ], + "8": [ + 138 + ], + "9": [ + 139 + ], + ":": [ + 11 + ], + ";": [ + 12 + ], + "?": [ + 13 + ], + "X": [ + 156 + ], + "^": [ + 1 + ], + "_": [ + 0 + ], + "a": [ + 14 + ], + "b": [ + 15 + ], + "c": [ + 16 + ], + "d": [ + 17 + ], + "e": [ + 18 + ], + "f": [ + 19 + ], + "g": [ + 154 + ], + "h": [ + 20 + ], + "i": [ + 21 + ], + "j": [ + 22 + ], + "k": [ + 23 + ], + "l": [ + 24 + ], + "m": [ + 25 + ], + "n": [ + 26 + ], + "o": [ + 27 + ], + "p": [ + 28 + ], + "q": [ + 29 + ], + "r": [ + 30 + ], + "s": [ + 31 + ], + "t": [ + 32 + ], + "u": [ + 33 + ], + "v": [ + 34 + ], + "w": [ + 35 + ], + "x": [ + 36 + ], + "y": [ + 37 + ], + "z": [ + 38 + ], + "æ": [ + 39 + ], + "ç": [ + 40 + ], + "ð": [ + 41 + ], + "ø": [ + 42 + ], + "ħ": [ + 43 + ], + "ŋ": [ + 44 + ], + "œ": [ + 45 + ], + "ǀ": [ + 46 + ], + "ǁ": [ + 47 + ], + "ǂ": [ + 48 + ], + "ǃ": [ + 49 + ], + "ɐ": [ + 50 + ], + "ɑ": [ + 51 + ], + "ɒ": [ + 52 + ], + "ɓ": [ + 53 + ], + "ɔ": [ + 54 + ], + "ɕ": [ + 55 + ], + "ɖ": [ + 56 + ], + "ɗ": [ + 57 + ], + "ɘ": [ + 58 + ], + "ə": [ + 59 + ], + "ɚ": [ + 60 + ], + "ɛ": [ + 61 + ], + "ɜ": [ + 62 + ], + "ɞ": [ + 63 + ], + "ɟ": [ + 64 + ], + "ɠ": [ + 65 + ], + "ɡ": [ + 66 + ], + "ɢ": [ + 67 + ], + "ɣ": [ + 68 + ], + "ɤ": [ + 69 + ], + "ɥ": [ + 70 + ], + "ɦ": [ + 71 + ], + "ɧ": [ + 72 + ], + "ɨ": [ + 73 + ], + "ɪ": [ + 74 + ], + "ɫ": [ + 75 + ], + "ɬ": [ + 76 + ], + "ɭ": [ + 77 + ], + "ɮ": [ + 78 + ], + "ɯ": [ + 79 + ], + "ɰ": [ + 80 + ], + "ɱ": [ + 81 + ], + "ɲ": [ + 82 + ], + "ɳ": [ + 83 + ], + "ɴ": [ + 84 + ], + "ɵ": [ + 85 + ], + "ɶ": [ + 86 + ], + "ɸ": [ + 87 + ], + "ɹ": [ + 88 + ], + "ɺ": [ + 89 + ], + "ɻ": [ + 90 + ], + "ɽ": [ + 91 + ], + "ɾ": [ + 92 + ], + "ʀ": [ + 93 + ], + "ʁ": [ + 94 + ], + "ʂ": [ + 95 + ], + "ʃ": [ + 96 + ], + "ʄ": [ + 97 + ], + "ʈ": [ + 98 + ], + "ʉ": [ + 99 + ], + "ʊ": [ + 100 + ], + "ʋ": [ + 101 + ], + "ʌ": [ + 102 + ], + "ʍ": [ + 103 + ], + "ʎ": [ + 104 + ], + "ʏ": [ + 105 + ], + "ʐ": [ + 106 + ], + "ʑ": [ + 107 + ], + "ʒ": [ + 108 + ], + "ʔ": [ + 109 + ], + "ʕ": [ + 110 + ], + "ʘ": [ + 111 + ], + "ʙ": [ + 112 + ], + "ʛ": [ + 113 + ], + "ʜ": [ + 114 + ], + "ʝ": [ + 115 + ], + "ʟ": [ + 116 + ], + "ʡ": [ + 117 + ], + "ʢ": [ + 118 + ], + "ʦ": [ + 155 + ], + "ʰ": [ + 145 + ], + "ʲ": [ + 119 + ], + "ˈ": [ + 120 + ], + "ˌ": [ + 121 + ], + "ː": [ + 122 + ], + "ˑ": [ + 123 + ], + "˞": [ + 124 + ], + "ˤ": [ + 146 + ], + "̃": [ + 141 + ], + "̊": [ + 158 + ], + "̝": [ + 157 + ], + "̧": [ + 140 + ], + "̩": [ + 144 + ], + "̪": [ + 142 + ], + "̯": [ + 143 + ], + "̺": [ + 152 + ], + "̻": [ + 153 + ], + "β": [ + 125 + ], + "ε": [ + 147 + ], + "θ": [ + 126 + ], + "χ": [ + 127 + ], + "ᵻ": [ + 128 + ], + "↑": [ + 151 + ], + "↓": [ + 148 + ], + "ⱱ": [ + 129 + ] + }, + "num_symbols": 256, + "num_speakers": 52, + "speaker_id_map": { + "2450": 0, + "1724": 1, + "1666": 2, + "5809": 3, + "496": 4, + "2506": 5, + "7432": 6, + "3619": 7, + "4429": 8, + "3798": 9, + "12500": 10, + "10587": 11, + "2951": 12, + "1775": 13, + "9861": 14, + "880": 15, + "3034": 16, + "2825": 17, + "5438": 18, + "3245": 19, + "4396": 20, + "11290": 21, + "11936": 22, + "6916": 23, + "10294": 24, + "10079": 25, + "7588": 26, + "7579": 27, + "123": 28, + "3024": 29, + "960": 30, + "10984": 31, + "2792": 32, + "7723": 33, + "4174": 34, + "2981": 35, + "5764": 36, + "6513": 37, + "7884": 38, + "6697": 39, + "12749": 40, + "11157": 41, + "2239": 42, + "10879": 43, + "1085": 44, + "8480": 45, + "8331": 46, + "6282": 47, + "10632": 48, + "2602": 49, + "5367": 50, + "11472": 51 + }, + "piper_version": "1.0.0" +} From 2dbff77c61d023622b1f5450205b6e95cb34021e Mon Sep 17 00:00:00 2001 From: Michael Hansen Date: Mon, 5 Feb 2024 14:07:29 -0600 Subject: [PATCH 02/16] Add --min-phoneme-count --- README.md | 33 +++++++++++ generate_samples.py | 139 ++++++++++++++++++++++++++++++++------------ pylintrc | 7 +++ requirements.txt | 2 +- 4 files changed, 143 insertions(+), 38 deletions(-) diff --git a/README.md b/README.md index 3558195..b89b413 100644 --- a/README.md +++ b/README.md @@ -2,6 +2,13 @@ Generates samples using [Piper](https://github.com/rhasspy/piper/) for training a wake word system like [openWakeWord](https://github.com/dscripka/openWakeWord). +Available models: + +* [English](https://github.com/rhasspy/piper-sample-generator/releases/download/v2.0.0/en_US-libritts_r-medium.pt) +* [French](https://github.com/rhasspy/piper-sample-generator/releases/download/v2.0.0/fr_FR-mls-medium.pt) +* [German](https://github.com/rhasspy/piper-sample-generator/releases/download/v2.0.0/de_DE-mls-medium.pt) +* [Dutch](https://github.com/rhasspy/piper-sample-generator/releases/download/v2.0.0/nl_NL-mls-medium.pt) + ## Install @@ -23,6 +30,7 @@ Download the LibriTTS-R generator (exported from [checkpoint](https://huggingfac wget -O models/en-us-libritts-high.pt 'https://github.com/rhasspy/piper-sample-generator/releases/download/v2.0.0/en_US-libritts_r-medium.pt' ``` +See links above for models for other languages. ## Run @@ -72,3 +80,28 @@ This will do several things to each sample: * Change the acoustics of the sample to sound like the speaker was in a room with echo or using a poor quality microphone 3. Resample to 16Khz for training (e.g., [openWakeWord](https://github.com/dscripka/openWakeWord)) + +## Short Phrases + +Models that were trained on audio books tend to perform poorly when speaking short phrases or single words. +The French, German, and Dutch models trained from the [MLS](http://openslr.org/94/) have this problem. + +The problem can be mitigated by repeating the phrase over and over, and then clipping out a single sample. +To do this automatically, follow these steps: + +1. Ensure your short phrase ends with a comma (`,`) +2. Lower the noise settings with `--noise-scales 0.333` and `--noise-scale-ws 0.333` +3. Use `--min-phoneme-count 300` (the value 300 was determined empirically and may be less for some models) + +For example: + +``` sh +python3 generate_samples.py \ + 'framboise,' \ + --model models/fr_FR-mls-medium.pt \ + --noise-scales 0.333 \ + --noise-scale-ws 0.333 \ + --min-phoneme-count 300 + --max-samples 1 \ + --output-dir . +``` diff --git a/generate_samples.py b/generate_samples.py index 349ebde..d0a6e85 100755 --- a/generate_samples.py +++ b/generate_samples.py @@ -7,7 +7,7 @@ import logging import os import wave from pathlib import Path -from typing import List, Union +from typing import Any, Dict, List, Optional, Tuple, Union import numpy as np import torch @@ -24,21 +24,20 @@ logging.basicConfig(level=logging.DEBUG) # Main generation function def generate_samples( - text: Union[List, str], - output_dir: str, - max_samples: int = None, - file_names: List[str] = [], - model: str = os.path.join( - Path(__file__).parent, "models", "en_US-libritts_r-medium.pt" - ), + text: Union[List[str], str], + output_dir: Union[str, Path], + max_samples: Optional[int] = None, + file_names: Optional[List[str]] = None, + model: Union[str, Path] = _DIR / "models" / "en_US-libritts_r-medium.pt", batch_size: int = 1, - slerp_weights: List[float] = [0.5], - length_scales: List[float] = [0.75, 1, 1.25], - noise_scales: List[float] = [0.667], - noise_scale_ws: List[float] = [0.8], - max_speakers: float = None, + slerp_weights: Tuple[float, ...] = (0.5,), + length_scales: Tuple[float, ...] = (0.75, 1, 1.25), + noise_scales: Tuple[float, ...] = (0.667,), + noise_scale_ws: Tuple[float, ...] = (0.8,), + max_speakers: Optional[float] = None, verbose: bool = False, auto_reduce_batch_size: bool = False, + min_phoneme_count: Optional[int] = None, **kwargs, ) -> None: """ @@ -61,6 +60,8 @@ def generate_samples( verbose (bool): Enable or disable more detailed logging messages (default: False). auto_reduce_batch_size (bool): Automatically and temporarily reduce the batch size if CUDA OOM errors are detected, and try to resume generation. + min_phoneme_count (int): If set, ensure this number of phonemes is always sent to the model. + Clip audio to extract original phrase. Returns: None @@ -71,12 +72,13 @@ def generate_samples( _LOGGER.debug("Loading %s", model) model_path = Path(model) - model = torch.load(model_path) - model.eval() + + torch_model = torch.load(model_path) + torch_model.eval() _LOGGER.info("Successfully loaded the model") if torch.cuda.is_available(): - model.cuda() + torch_model.cuda() _LOGGER.debug("CUDA available, using GPU") output_dir = Path(output_dir) @@ -147,10 +149,14 @@ def generate_samples( speaker_1 = torch.LongTensor([s[0] for s in speakers_batch]) speaker_2 = torch.LongTensor([s[1] for s in speakers_batch]) - phoneme_ids = [ - get_phonemes(voice, config, next(texts), verbose) - for i in range(batch_size) - ] + phoneme_ids_by_batch = [] + clip_indexes_by_batch = [] + for i in range(batch_size): + phoneme_ids, clip_phoneme_index = get_phonemes( + voice, config, next(texts), verbose, min_phoneme_count + ) + phoneme_ids_by_batch.append(phoneme_ids) + clip_indexes_by_batch.append(clip_phoneme_index) def right_pad_lists(lists): max_length = max(len(lst) for lst in lists) @@ -162,18 +168,18 @@ def generate_samples( padded_lists.append(padded_l) return padded_lists - phoneme_ids = right_pad_lists(phoneme_ids) + phoneme_ids_by_batch = right_pad_lists(phoneme_ids_by_batch) if auto_reduce_batch_size: oom_error = True counter = 1 while oom_error is True: try: - audio = generate_audio( - model, + audio, phoneme_samples = generate_audio( + torch_model, speaker_1[0 : batch_size // counter], speaker_2[0 : batch_size // counter], - phoneme_ids[0 : batch_size // counter], + phoneme_ids_by_batch[0 : batch_size // counter], slerp_weight, noise_scale, noise_scale_w, @@ -186,11 +192,11 @@ def generate_samples( gc.collect() counter += 1 # reduce batch size to avoid OOM errors else: - audio = generate_audio( - model, + audio, phoneme_samples = generate_audio( + torch_model, speaker_1, speaker_2, - phoneme_ids, + phoneme_ids_by_batch, slerp_weight, noise_scale, noise_scale_w, @@ -198,19 +204,34 @@ def generate_samples( max_len, ) + # Clip audio when using min_phoneme_count + for i, clip_phoneme_index in enumerate(clip_indexes_by_batch): + if clip_phoneme_index is not None: + last_sample_idx = int( + phoneme_samples[i].flatten()[clip_phoneme_index:].sum().item() + ) + + # Fill remainder of audio with silence. + # It will be removed in the next stage. + audio[i, 0, :-last_sample_idx] = 0 + # Resample audio audio = resampler(audio.cpu()).numpy() audio_int16 = audio_float_to_int16(audio) for audio_idx in range(audio_int16.shape[0]): # Use webrtcvad to trip silence from the clips - audio_data = remove_silence(audio_int16[audio_idx].flatten())[None,] + audio_data = remove_silence(audio_int16[audio_idx].flatten())[ + None, + ] if isinstance(file_names, it.cycle): wav_path = output_dir / next(file_names) else: wav_path = output_dir / f"{sample_idx}.wav" - with wave.open(str(wav_path), "wb") as wav_file: + + wav_file: wave.Wave_write = wave.open(str(wav_path), "wb") + with wav_file: wav_file.setframerate(resample_rate) wav_file.setsampwidth(2) wav_file.setnchannels(1) @@ -224,17 +245,22 @@ def generate_samples( # print(f"Batch {batch_idx +1}/{max_samples//batch_size} complete", " "*200, end='\r') # Next batch - _LOGGER.debug(f"Batch {batch_idx +1}/{max_samples//batch_size} complete") + _LOGGER.debug("Batch %s/%s complete", batch_idx + 1, max_samples // batch_size) speakers_batch = list(it.islice(speakers_iter, 0, batch_size)) batch_idx += 1 _LOGGER.info("Done") -def remove_silence(x, frame_duration=0.030, sample_rate=16000, min_start=2000): +def remove_silence( + x: np.ndarray, + frame_duration: float = 0.030, + sample_rate: int = 16000, + min_start: int = 2000, +) -> np.ndarray: """Uses webrtc voice activity detection to remove silence from the clips""" vad = webrtcvad.Vad(0) - if x.dtype == np.float32 or x.dtype == np.float64: + if x.dtype in (np.float32, np.float64): x = (x * 32767).astype(np.int16) x_new = x[0:min_start].tolist() step_size = int(sample_rate * frame_duration) @@ -295,10 +321,18 @@ def generate_audio( o = model.dec((z * y_mask)[:, :, :max_len], g=g) audio = o - return audio + phoneme_samples = w_ceil * 256 # hop length + + return audio, phoneme_samples -def get_phonemes(voice, config, text, verbose): +def get_phonemes( + voice: str, + config: Dict[str, Any], + text: str, + verbose: bool = False, + min_phoneme_count: Optional[int] = None, +) -> Tuple[List[int], Optional[int]]: # Combine all sentences phonemes = [ p @@ -309,18 +343,41 @@ def get_phonemes(voice, config, text, verbose): _LOGGER.debug("Phonemes: %s", phonemes) id_map = config["phoneme_id_map"] + + # Beginning of utterance phoneme_ids = list(id_map["^"]) + + # Phoneme ids for just the text + text_phoneme_ids = [] + for phoneme in phonemes: p_ids = id_map.get(phoneme) if p_ids is not None: phoneme_ids.extend(p_ids) + text_phoneme_ids.extend(p_ids) phoneme_ids.extend(id_map["_"]) + text_phoneme_ids.extend(id_map["_"]) + # Index where audio should be clipped at. + # When None, all of the audio will be used. + clip_phoneme_index: Optional[int] = None + + if min_phoneme_count is not None: + # Repeat phrase until minimum phoneme count is met. + # NOTE: It is critical that the ^ and $ phonemes are not repeated here. + while (len(phoneme_ids) - 1) < min_phoneme_count: + # We will clip audio at the beginning of the last phrase + clip_phoneme_index = len(phoneme_ids) - 1 + + phoneme_ids.extend(text_phoneme_ids) + + # End of utterance phoneme_ids.extend(id_map["$"]) - return phoneme_ids + + return phoneme_ids, clip_phoneme_index -def slerp(v1, v2, t, DOT_THR=0.9995, zdim=-1): +def slerp(v1, v2, t: float, DOT_THR: float = 0.9995, zdim: int = -1): """SLERP for pytorch tensors interpolating `v1` to `v2` with scale of `t`. `DOT_THR` determines when the vectors are too close to parallel. @@ -375,7 +432,9 @@ def audio_float_to_int16( return audio_norm -if __name__ == "__main__": +def main() -> None: + """Main entry point.""" + # Get command line arguments parser = argparse.ArgumentParser() parser.add_argument("text") @@ -401,7 +460,13 @@ if __name__ == "__main__": type=int, help="Maximum number of speakers to use (default: all)", ) + parser.add_argument("--min-phoneme-count", type=int) + parser.add_argument("--verbose", action="store_true") args = parser.parse_args().__dict__ # Generate speech generate_samples(**args) + + +if __name__ == "__main__": + main() diff --git a/pylintrc b/pylintrc index 22a70d0..561b2f1 100644 --- a/pylintrc +++ b/pylintrc @@ -35,3 +35,10 @@ disable= [FORMAT] expected-line-ending-format=LF + +[TYPECHECK] + +# List of members which are set dynamically and missed by pylint inference +# system, and so shouldn't trigger E1101 when accessed. Python regular +# expressions are accepted. +generated-members=numpy.*,torch.* diff --git a/requirements.txt b/requirements.txt index c6ac75a..8443d84 100644 --- a/requirements.txt +++ b/requirements.txt @@ -1,6 +1,6 @@ audiomentations==0.33.0 piper-phonemize==1.1.0 numpy<2 -torch +torch<2 torchaudio webrtcvad From 315e555f4962e65b7b6404955bd57e317e5e507c Mon Sep 17 00:00:00 2001 From: Michael Hansen Date: Mon, 5 Feb 2024 14:18:15 -0600 Subject: [PATCH 03/16] Add pad after bos --- generate_samples.py | 1 + 1 file changed, 1 insertion(+) diff --git a/generate_samples.py b/generate_samples.py index d0a6e85..46ecedb 100755 --- a/generate_samples.py +++ b/generate_samples.py @@ -346,6 +346,7 @@ def get_phonemes( # Beginning of utterance phoneme_ids = list(id_map["^"]) + phoneme_ids.extend(id_map["_"]) # Phoneme ids for just the text text_phoneme_ids = [] From 172d7b5cae6210cac9e4cb685d183f61148fb1df Mon Sep 17 00:00:00 2001 From: Kevin Ahrendt Date: Sat, 24 Feb 2024 16:03:56 -0500 Subject: [PATCH 04/16] use phoneme lengths to trim --- generate_samples.py | 41 +++++++++++------------------------------ requirements.txt | 1 - 2 files changed, 11 insertions(+), 31 deletions(-) diff --git a/generate_samples.py b/generate_samples.py index 46ecedb..b31da1c 100755 --- a/generate_samples.py +++ b/generate_samples.py @@ -12,7 +12,6 @@ from typing import Any, Dict, List, Optional, Tuple, Union import numpy as np import torch import torchaudio -import webrtcvad from piper_phonemize import phonemize_espeak from piper_train.vits import commons @@ -115,7 +114,7 @@ def generate_samples( resample_rate, lowpass_filter_width=64, rolloff=0.9475937167399596, - resampling_method="kaiser_window", + resampling_method="sinc_interp_kaiser", beta=14.769656459379492, ) @@ -207,23 +206,25 @@ def generate_samples( # Clip audio when using min_phoneme_count for i, clip_phoneme_index in enumerate(clip_indexes_by_batch): if clip_phoneme_index is not None: - last_sample_idx = int( - phoneme_samples[i].flatten()[clip_phoneme_index:].sum().item() + first_sample_idx = int( + phoneme_samples[i].flatten()[:clip_phoneme_index-1].sum().item() ) - + # Fill remainder of audio with silence. # It will be removed in the next stage. - audio[i, 0, :-last_sample_idx] = 0 + audio[i, 0, :first_sample_idx] = 0 + + # Fill time after last speech with silence. + # It will be removed in the next stage + last_sample_idx = int(phoneme_samples[i].flatten().sum().item()) + audio[i, 0, last_sample_idx+1:] = 0 # Resample audio audio = resampler(audio.cpu()).numpy() audio_int16 = audio_float_to_int16(audio) for audio_idx in range(audio_int16.shape[0]): - # Use webrtcvad to trip silence from the clips - audio_data = remove_silence(audio_int16[audio_idx].flatten())[ - None, - ] + audio_data = np.trim_zeros(audio_int16[audio_idx].flatten()) if isinstance(file_names, it.cycle): wav_path = output_dir / next(file_names) @@ -251,26 +252,6 @@ def generate_samples( _LOGGER.info("Done") - -def remove_silence( - x: np.ndarray, - frame_duration: float = 0.030, - sample_rate: int = 16000, - min_start: int = 2000, -) -> np.ndarray: - """Uses webrtc voice activity detection to remove silence from the clips""" - vad = webrtcvad.Vad(0) - if x.dtype in (np.float32, np.float64): - x = (x * 32767).astype(np.int16) - x_new = x[0:min_start].tolist() - step_size = int(sample_rate * frame_duration) - for i in range(min_start, x.shape[0] - step_size, step_size): - vad_res = vad.is_speech(x[i : i + step_size].tobytes(), sample_rate) - if vad_res: - x_new.extend(x[i : i + step_size].tolist()) - return np.array(x_new).astype(np.int16) - - def generate_audio( model, speaker_1, diff --git a/requirements.txt b/requirements.txt index 8443d84..6f93d3d 100644 --- a/requirements.txt +++ b/requirements.txt @@ -3,4 +3,3 @@ piper-phonemize==1.1.0 numpy<2 torch<2 torchaudio -webrtcvad From 213d4d561ab8a84f71de7dddac827cb07e92c031 Mon Sep 17 00:00:00 2001 From: Kevin Ahrendt Date: Tue, 27 Feb 2024 10:01:27 -0500 Subject: [PATCH 05/16] revert removing webrtcvad --- generate_samples.py | 29 ++++++++++++++++++++++++++++- requirements.txt | 1 + 2 files changed, 29 insertions(+), 1 deletion(-) diff --git a/generate_samples.py b/generate_samples.py index b31da1c..bb582bc 100755 --- a/generate_samples.py +++ b/generate_samples.py @@ -12,6 +12,7 @@ from typing import Any, Dict, List, Optional, Tuple, Union import numpy as np import torch import torchaudio +import webrtcvad from piper_phonemize import phonemize_espeak from piper_train.vits import commons @@ -210,7 +211,7 @@ def generate_samples( phoneme_samples[i].flatten()[:clip_phoneme_index-1].sum().item() ) - # Fill remainder of audio with silence. + # Fill start of audio with silence until actual sample. # It will be removed in the next stage. audio[i, 0, :first_sample_idx] = 0 @@ -224,7 +225,13 @@ def generate_samples( audio_int16 = audio_float_to_int16(audio) for audio_idx in range(audio_int16.shape[0]): + # Trim any silenced audio audio_data = np.trim_zeros(audio_int16[audio_idx].flatten()) + + # Use webrtcvad to trim any remaining silence from the clips + audio_data = remove_silence(audio_int16[audio_idx].flatten())[ + None, + ] if isinstance(file_names, it.cycle): wav_path = output_dir / next(file_names) @@ -252,6 +259,26 @@ def generate_samples( _LOGGER.info("Done") + +def remove_silence( + x: np.ndarray, + frame_duration: float = 0.030, + sample_rate: int = 16000, + min_start: int = 2000, +) -> np.ndarray: + """Uses webrtc voice activity detection to remove silence from the clips""" + vad = webrtcvad.Vad(0) + if x.dtype in (np.float32, np.float64): + x = (x * 32767).astype(np.int16) + x_new = x[0:min_start].tolist() + step_size = int(sample_rate * frame_duration) + for i in range(min_start, x.shape[0] - step_size, step_size): + vad_res = vad.is_speech(x[i : i + step_size].tobytes(), sample_rate) + if vad_res: + x_new.extend(x[i : i + step_size].tolist()) + return np.array(x_new).astype(np.int16) + + def generate_audio( model, speaker_1, diff --git a/requirements.txt b/requirements.txt index 6f93d3d..8443d84 100644 --- a/requirements.txt +++ b/requirements.txt @@ -3,3 +3,4 @@ piper-phonemize==1.1.0 numpy<2 torch<2 torchaudio +webrtcvad From 4057c1a620e9f644a598d7aaf524c8eeb494fc41 Mon Sep 17 00:00:00 2001 From: Michael Hansen Date: Fri, 29 Aug 2025 12:19:25 -0500 Subject: [PATCH 06/16] Upgrade to torch 2, piper 1.3 --- CHANGELOG.md | 9 ++ generate_samples.py | 267 ++++++++++++++++++++++++-------------------- pyproject.toml | 52 +++++++++ script/format | 11 +- script/lint | 17 ++- script/run | 9 +- script/setup | 12 +- 7 files changed, 243 insertions(+), 134 deletions(-) create mode 100644 CHANGELOG.md create mode 100644 pyproject.toml diff --git a/CHANGELOG.md b/CHANGELOG.md new file mode 100644 index 0000000..d3bb09a --- /dev/null +++ b/CHANGELOG.md @@ -0,0 +1,9 @@ +# Changelog + +## 3.0.0 + +- Move phonemization to piper 1.3.0 (piper-phonemize is deprecated) +- Move to PyTorch 2 +- Add support for using Piper voices (`.onnx`) directly +- Remove silence trimming +- Remove `min-phoneme-count` diff --git a/generate_samples.py b/generate_samples.py index bb582bc..c12acaf 100755 --- a/generate_samples.py +++ b/generate_samples.py @@ -1,19 +1,19 @@ #!/usr/bin/env python3 import argparse -import gc import itertools as it import json import logging import os import wave +from collections.abc import Iterable from pathlib import Path -from typing import Any, Dict, List, Optional, Tuple, Union +from typing import Any, Dict, List, Optional, Tuple, Union, cast import numpy as np import torch import torchaudio -import webrtcvad -from piper_phonemize import phonemize_espeak +from piper import PiperVoice, SynthesisConfig +from piper.phonemize_espeak import EspeakPhonemizer from piper_train.vits import commons @@ -27,17 +27,15 @@ def generate_samples( text: Union[List[str], str], output_dir: Union[str, Path], max_samples: Optional[int] = None, - file_names: Optional[List[str]] = None, + file_names: Optional[Iterable[str]] = None, model: Union[str, Path] = _DIR / "models" / "en_US-libritts_r-medium.pt", batch_size: int = 1, slerp_weights: Tuple[float, ...] = (0.5,), length_scales: Tuple[float, ...] = (0.75, 1, 1.25), noise_scales: Tuple[float, ...] = (0.667,), noise_scale_ws: Tuple[float, ...] = (0.8,), - max_speakers: Optional[float] = None, + max_speakers: Optional[int] = None, verbose: bool = False, - auto_reduce_batch_size: bool = False, - min_phoneme_count: Optional[int] = None, **kwargs, ) -> None: """ @@ -50,7 +48,7 @@ def generate_samples( max_samples (int): The maximum number of samples to generate. file_names (List[str]): The names to use when saving the files. Must be the same length as the `text` argument, if a list. - model (str): The path to the STT model to use for generation. + model (str): The path to the TTS generator model (.pt). batch_size (int): The batch size to use when generated the clips slerp_weights (List[float]): The weights to use when mixing speakers via SLERP. length_scales (List[float]): Controls the average duration/speed of the generated speech. @@ -58,11 +56,6 @@ def generate_samples( noise_scale_ws (List[float]): A parameter for the stochastic duration of words/phonemes. max_speakers (int): The maximum speaker number to use, if the model is multi-speaker. verbose (bool): Enable or disable more detailed logging messages (default: False). - auto_reduce_batch_size (bool): Automatically and temporarily reduce the batch size - if CUDA OOM errors are detected, and try to resume generation. - min_phoneme_count (int): If set, ensure this number of phonemes is always sent to the model. - Clip audio to extract original phrase. - Returns: None """ @@ -73,7 +66,7 @@ def generate_samples( _LOGGER.debug("Loading %s", model) model_path = Path(model) - torch_model = torch.load(model_path) + torch_model = torch.load(model_path, weights_only=False) torch_model.eval() _LOGGER.info("Successfully loaded the model") @@ -108,14 +101,13 @@ def generate_samples( ) # Define resampler to get to 16khz (https://pytorch.org/audio/stable/tutorials/audio_resampling_tutorial.html#kaiser-best) - sample_rate = 22050 resample_rate = 16000 resampler = torchaudio.transforms.Resample( sample_rate, resample_rate, lowpass_filter_width=64, rolloff=0.9475937167399596, - resampling_method="sinc_interp_kaiser", + # resampling_method="sinc_interp_kaiser", beta=14.769656459379492, ) @@ -150,13 +142,9 @@ def generate_samples( speaker_2 = torch.LongTensor([s[1] for s in speakers_batch]) phoneme_ids_by_batch = [] - clip_indexes_by_batch = [] for i in range(batch_size): - phoneme_ids, clip_phoneme_index = get_phonemes( - voice, config, next(texts), verbose, min_phoneme_count - ) + phoneme_ids = get_phonemes(voice, config, next(texts), verbose) phoneme_ids_by_batch.append(phoneme_ids) - clip_indexes_by_batch.append(clip_phoneme_index) def right_pad_lists(lists): max_length = max(len(lst) for lst in lists) @@ -169,69 +157,24 @@ def generate_samples( return padded_lists phoneme_ids_by_batch = right_pad_lists(phoneme_ids_by_batch) - - if auto_reduce_batch_size: - oom_error = True - counter = 1 - while oom_error is True: - try: - audio, phoneme_samples = generate_audio( - torch_model, - speaker_1[0 : batch_size // counter], - speaker_2[0 : batch_size // counter], - phoneme_ids_by_batch[0 : batch_size // counter], - slerp_weight, - noise_scale, - noise_scale_w, - length_scale, - max_len, - ) - oom_error = False - except torch.cuda.OutOfMemoryError: - torch.cuda.empty_cache() - gc.collect() - counter += 1 # reduce batch size to avoid OOM errors - else: - audio, phoneme_samples = generate_audio( - torch_model, - speaker_1, - speaker_2, - phoneme_ids_by_batch, - slerp_weight, - noise_scale, - noise_scale_w, - length_scale, - max_len, - ) - - # Clip audio when using min_phoneme_count - for i, clip_phoneme_index in enumerate(clip_indexes_by_batch): - if clip_phoneme_index is not None: - first_sample_idx = int( - phoneme_samples[i].flatten()[:clip_phoneme_index-1].sum().item() - ) - - # Fill start of audio with silence until actual sample. - # It will be removed in the next stage. - audio[i, 0, :first_sample_idx] = 0 - - # Fill time after last speech with silence. - # It will be removed in the next stage - last_sample_idx = int(phoneme_samples[i].flatten().sum().item()) - audio[i, 0, last_sample_idx+1:] = 0 + audio = generate_audio( + torch_model, + speaker_1, + speaker_2, + phoneme_ids_by_batch, + slerp_weight, + noise_scale, + noise_scale_w, + length_scale, + max_len, + ) # Resample audio - audio = resampler(audio.cpu()).numpy() + audio_np = resampler(audio.cpu()).numpy() - audio_int16 = audio_float_to_int16(audio) + audio_int16 = audio_float_to_int16(audio_np) for audio_idx in range(audio_int16.shape[0]): - # Trim any silenced audio audio_data = np.trim_zeros(audio_int16[audio_idx].flatten()) - - # Use webrtcvad to trim any remaining silence from the clips - audio_data = remove_silence(audio_int16[audio_idx].flatten())[ - None, - ] if isinstance(file_names, it.cycle): wav_path = output_dir / next(file_names) @@ -260,23 +203,107 @@ def generate_samples( _LOGGER.info("Done") -def remove_silence( - x: np.ndarray, - frame_duration: float = 0.030, - sample_rate: int = 16000, - min_start: int = 2000, -) -> np.ndarray: - """Uses webrtc voice activity detection to remove silence from the clips""" - vad = webrtcvad.Vad(0) - if x.dtype in (np.float32, np.float64): - x = (x * 32767).astype(np.int16) - x_new = x[0:min_start].tolist() - step_size = int(sample_rate * frame_duration) - for i in range(min_start, x.shape[0] - step_size, step_size): - vad_res = vad.is_speech(x[i : i + step_size].tobytes(), sample_rate) - if vad_res: - x_new.extend(x[i : i + step_size].tolist()) - return np.array(x_new).astype(np.int16) +# ----------------------------------------------------------------------------- + + +def generate_samples_onnx( + text: Union[List[str], str], + output_dir: Union[str, Path], + model: Union[str, Path], + max_samples: Optional[int] = None, + file_names: Optional[Iterable[str]] = None, + length_scales: Tuple[float, ...] = (0.75, 1, 1.25), + noise_scales: Tuple[float, ...] = (0.667,), + noise_scale_ws: Tuple[float, ...] = (0.8,), + max_speakers: Optional[int] = None, + **kwargs, +) -> None: + """ + Generate synthetic speech clips, saving the clips to the specified output directory. + + Args: + text (List[str]): The text to convert into speech. Can be either a + a list of strings, or a path to a file with text on each line. + output_dir (str): The location to save the generated clips. + model (str): The path to the Piper TTS model (.onnx). + max_samples (int): The maximum number of samples to generate. + file_names (List[str]): The names to use when saving the files. Must be the same length + as the `text` argument, if a list. + length_scales (List[float]): Controls the average duration/speed of the generated speech. + noise_scales (List[float]): A parameter for overall variability of the generated speech. + noise_scale_ws (List[float]): A parameter for the stochastic duration of words/phonemes. + max_speakers (int): The maximum speaker number to use, if the model is multi-speaker. + + Returns: + None + """ + + if max_samples is None: + max_samples = len(text) + + _LOGGER.debug("Loading %s", model) + voice = PiperVoice.load(model, use_cuda=torch.cuda.is_available()) + _LOGGER.info("Successfully loaded the model") + + output_dir = Path(output_dir) + output_dir.mkdir(parents=True, exist_ok=True) + + num_speakers = voice.config.num_speakers + if max_speakers is not None: + num_speakers = min(num_speakers, max_speakers) + + sample_idx = 0 + settings_iter = it.cycle( + it.product( + list(range(num_speakers)), + length_scales, + noise_scales, + noise_scale_ws, + ) + ) + + if isinstance(text, str) and os.path.exists(text): + texts = it.cycle( + [ + i.strip() + for i in open(text, "r", encoding="utf-8").readlines() + if len(i.strip()) > 0 + ] + ) + elif isinstance(text, list): + texts = it.cycle(text) + else: + texts = it.cycle([text]) + + if file_names: + file_names = it.cycle(file_names) + + for speaker_id, length_scale, noise_scale, noise_w_scale in settings_iter: + if isinstance(file_names, it.cycle): + wav_path = output_dir / next(file_names) + else: + wav_path = output_dir / f"{sample_idx}.wav" + + wav_file: wave.Wave_write = wave.open(str(wav_path), "wb") + voice.synthesize_wav( + next(texts), + wav_file=wav_file, + syn_config=SynthesisConfig( + speaker_id=speaker_id, + length_scale=length_scale, + noise_scale=noise_scale, + noise_w_scale=noise_w_scale, + ), + ) + + sample_idx += 1 + if sample_idx >= max_samples: + break + + _LOGGER.info("Done") + + +# ----------------------------------------------------------------------------- def generate_audio( @@ -289,15 +316,15 @@ def generate_audio( noise_scale_w, length_scale, max_len, -): +) -> torch.FloatTensor: x = torch.LongTensor(phoneme_ids) x_lengths = torch.LongTensor([len(i) for i in phoneme_ids]) if torch.cuda.is_available(): speaker_1 = speaker_1.cuda() speaker_2 = speaker_2.cuda() - x = x.cuda() - x_lengths = x_lengths.cuda() + x = cast(torch.LongTensor, x.cuda()) + x_lengths = cast(torch.LongTensor, x_lengths.cuda()) x, m_p_orig, logs_p_orig, x_mask = model.enc_p(x, x_lengths) emb0 = model.emb_g(speaker_1) @@ -312,7 +339,7 @@ def generate_audio( w_ceil = torch.ceil(w) y_lengths = torch.clamp_min(torch.sum(w_ceil, [1, 2]), 1).long() y_mask = torch.unsqueeze( - commons.sequence_mask(y_lengths, y_lengths.max()), 1 + commons.sequence_mask(y_lengths, int(y_lengths.max().item())), 1 ).type_as(x_mask) attn_mask = torch.unsqueeze(x_mask, 2) * torch.unsqueeze(y_mask, -1) attn = commons.generate_path(w_ceil, attn_mask) @@ -329,9 +356,11 @@ def generate_audio( o = model.dec((z * y_mask)[:, :, :max_len], g=g) audio = o - phoneme_samples = w_ceil * 256 # hop length - return audio, phoneme_samples + return audio + + +_PHONEMIZER = EspeakPhonemizer() def get_phonemes( @@ -339,12 +368,11 @@ def get_phonemes( config: Dict[str, Any], text: str, verbose: bool = False, - min_phoneme_count: Optional[int] = None, -) -> Tuple[List[int], Optional[int]]: +) -> List[int]: # Combine all sentences phonemes = [ p - for sentence_phonemes in phonemize_espeak(text, voice) + for sentence_phonemes in _PHONEMIZER.phonemize(voice, text) for p in sentence_phonemes ] if verbose is True: @@ -367,23 +395,10 @@ def get_phonemes( phoneme_ids.extend(id_map["_"]) text_phoneme_ids.extend(id_map["_"]) - # Index where audio should be clipped at. - # When None, all of the audio will be used. - clip_phoneme_index: Optional[int] = None - - if min_phoneme_count is not None: - # Repeat phrase until minimum phoneme count is met. - # NOTE: It is critical that the ^ and $ phonemes are not repeated here. - while (len(phoneme_ids) - 1) < min_phoneme_count: - # We will clip audio at the beginning of the last phrase - clip_phoneme_index = len(phoneme_ids) - 1 - - phoneme_ids.extend(text_phoneme_ids) - # End of utterance phoneme_ids.extend(id_map["$"]) - return phoneme_ids, clip_phoneme_index + return phoneme_ids def slerp(v1, v2, t: float, DOT_THR: float = 0.9995, zdim: int = -1): @@ -441,6 +456,9 @@ def audio_float_to_int16( return audio_norm +# ----------------------------------------------------------------------------- + + def main() -> None: """Main entry point.""" @@ -469,12 +487,17 @@ def main() -> None: type=int, help="Maximum number of speakers to use (default: all)", ) - parser.add_argument("--min-phoneme-count", type=int) parser.add_argument("--verbose", action="store_true") args = parser.parse_args().__dict__ # Generate speech - generate_samples(**args) + model_path = Path(args["model"]) + if model_path.suffix == ".onnx": + # Use Piper voice (.onnx) + generate_samples_onnx(**args) + else: + # Use PyTorch generator (.pt) + generate_samples(**args) if __name__ == "__main__": diff --git a/pyproject.toml b/pyproject.toml new file mode 100644 index 0000000..5061a05 --- /dev/null +++ b/pyproject.toml @@ -0,0 +1,52 @@ +[build-system] +requires = ["setuptools>=62.3"] +build-backend = "setuptools.build_meta" + +[project] +name = "piper-sample-generator" +version = "3.0.0" +license = {text = "Apache-2.0"} +description = "Generate TTS audio samples for training wake word systems" +readme = "README.md" +authors = [ + {name = "The Home Assistant Authors", email = "hello@home-assistant.io"} +] +keywords = ["piper", "sample", "tts", "wakeword"] +classifiers = [ + "Development Status :: 3 - Alpha", + "Intended Audience :: Developers", + "Topic :: Text Processing :: Linguistic", + "License :: OSI Approved :: Apache Software License", + "Programming Language :: Python :: 3.9", + "Programming Language :: Python :: 3.10", + "Programming Language :: Python :: 3.11", + "Programming Language :: Python :: 3.12", + "Programming Language :: Python :: 3.13", +] +requires-python = ">=3.9.0" +dependencies = [ + "piper-tts>=1.3.0,<2", + "torch>=2,<3", + "torchaudio", + "audiomentations", + "numpy", +] + +[project.optional-dependencies] +dev = [ + "black==24.8.0", + "flake8==7.2.0", + "mypy==1.14.0", + "pylint==3.2.7", + "pytest==8.3.5", +] + +[project.urls] +"Source Code" = "http://github.com/rhasspy/piper-sample-generator" + +[tool.setuptools] +platforms = ["any"] +zip-safe = true + +[tool.setuptools.packages.find] +include = [] diff --git a/script/format b/script/format index 19fa1f7..7f04417 100755 --- a/script/format +++ b/script/format @@ -8,6 +8,11 @@ _PROGRAM_DIR = _DIR.parent _VENV_DIR = _PROGRAM_DIR / ".venv" _SCRIPT = _PROGRAM_DIR / "generate_samples.py" -context = venv.EnvBuilder().ensure_directories(_VENV_DIR) -subprocess.check_call([context.env_exe, "-m", "black", str(_SCRIPT)]) -subprocess.check_call([context.env_exe, "-m", "isort", str(_SCRIPT)]) +if _VENV_DIR.exists(): + context = venv.EnvBuilder().ensure_directories(_VENV_DIR) + python_exe = context.env_exe +else: + python_exe = "python3" + +subprocess.check_call([python_exe, "-m", "black", str(_SCRIPT)]) +subprocess.check_call([python_exe, "-m", "isort", str(_SCRIPT)]) diff --git a/script/lint b/script/lint index d56932d..e4231e0 100755 --- a/script/lint +++ b/script/lint @@ -8,9 +8,14 @@ _PROGRAM_DIR = _DIR.parent _VENV_DIR = _PROGRAM_DIR / ".venv" _SCRIPT = _PROGRAM_DIR / "generate_samples.py" -context = venv.EnvBuilder().ensure_directories(_VENV_DIR) -subprocess.check_call([context.env_exe, "-m", "black", str(_SCRIPT), "--check"]) -subprocess.check_call([context.env_exe, "-m", "isort", str(_SCRIPT), "--check"]) -subprocess.check_call([context.env_exe, "-m", "flake8", str(_SCRIPT)]) -subprocess.check_call([context.env_exe, "-m", "pylint", str(_SCRIPT)]) -subprocess.check_call([context.env_exe, "-m", "mypy", str(_SCRIPT)]) +if _VENV_DIR.exists(): + context = venv.EnvBuilder().ensure_directories(_VENV_DIR) + python_exe = context.env_exe +else: + python_exe = "python3" + +subprocess.check_call([python_exe, "-m", "black", str(_SCRIPT), "--check"]) +subprocess.check_call([python_exe, "-m", "isort", str(_SCRIPT), "--check"]) +subprocess.check_call([python_exe, "-m", "flake8", str(_SCRIPT)]) +subprocess.check_call([python_exe, "-m", "pylint", str(_SCRIPT)]) +subprocess.check_call([python_exe, "-m", "mypy", str(_SCRIPT)]) diff --git a/script/run b/script/run index 5fa23ff..f2837d0 100755 --- a/script/run +++ b/script/run @@ -8,5 +8,10 @@ _DIR = Path(__file__).parent _PROGRAM_DIR = _DIR.parent _VENV_DIR = _PROGRAM_DIR / ".venv" -context = venv.EnvBuilder().ensure_directories(_VENV_DIR) -subprocess.check_call([context.env_exe, "generate_samples.py"] + sys.argv[1:]) +if _VENV_DIR.exists(): + context = venv.EnvBuilder().ensure_directories(_VENV_DIR) + python_exe = context.env_exe +else: + python_exe = "python3" + +subprocess.check_call([python_exe, "generate_samples.py"] + sys.argv[1:]) diff --git a/script/setup b/script/setup index 92ec185..0ff2e95 100755 --- a/script/setup +++ b/script/setup @@ -1,4 +1,5 @@ #!/usr/bin/env python3 +import argparse import subprocess import venv from pathlib import Path @@ -7,6 +8,9 @@ _DIR = Path(__file__).parent _PROGRAM_DIR = _DIR.parent _VENV_DIR = _PROGRAM_DIR / ".venv" +parser = argparse.ArgumentParser() +parser.add_argument("--dev", action="store_true", help="Install dev requirements") +args = parser.parse_args() # Create virtual environment builder = venv.EnvBuilder(with_pip=True) @@ -19,4 +23,10 @@ subprocess.check_call(pip + ["install", "--upgrade", "pip"]) subprocess.check_call(pip + ["install", "--upgrade", "setuptools", "wheel"]) # Install requirements -subprocess.check_call(pip + ["install", "-r", str(_PROGRAM_DIR / "requirements.txt")]) +subprocess.check_call(pip + ["install", "-e", str(_PROGRAM_DIR)]) + +if args.dev: + # Install dev requirements + subprocess.check_call( + pip + ["install", "-e", f"{_PROGRAM_DIR}[dev]"] + ) From 4d7e4b390c29bac54dd83e5f6688ef39ab12d578 Mon Sep 17 00:00:00 2001 From: Michael Hansen Date: Fri, 29 Aug 2025 15:16:20 -0500 Subject: [PATCH 07/16] Update README --- README.md | 101 ++++++++++++------------- augment.py | 2 +- generate_samples.py | 175 +++++++++++++++++++++++++++----------------- 3 files changed, 155 insertions(+), 123 deletions(-) diff --git a/README.md b/README.md index b89b413..fdc4117 100644 --- a/README.md +++ b/README.md @@ -1,14 +1,8 @@ # Piper Sample Generator -Generates samples using [Piper](https://github.com/rhasspy/piper/) for training a wake word system like [openWakeWord](https://github.com/dscripka/openWakeWord). - -Available models: - -* [English](https://github.com/rhasspy/piper-sample-generator/releases/download/v2.0.0/en_US-libritts_r-medium.pt) -* [French](https://github.com/rhasspy/piper-sample-generator/releases/download/v2.0.0/fr_FR-mls-medium.pt) -* [German](https://github.com/rhasspy/piper-sample-generator/releases/download/v2.0.0/de_DE-mls-medium.pt) -* [Dutch](https://github.com/rhasspy/piper-sample-generator/releases/download/v2.0.0/nl_NL-mls-medium.pt) +Generate spoken audio samples using [Piper][piper] for training a wake word system like [openWakeWord][] or [microWakeWord][]. +Supports normal [Piper voices][piper voices] or a special [generator][] that can mix speaker embeddings (English only). ## Install @@ -21,23 +15,45 @@ cd piper-sample-generator/ python3 -m venv .venv source .venv/bin/activate python3 -m pip install --upgrade pip -python3 -m pip install -r requirements.txt +python3 -m pip install -e . ``` -Download the LibriTTS-R generator (exported from [checkpoint](https://huggingface.co/datasets/rhasspy/piper-checkpoints/tree/main/en/en_US/libritts_r/medium)): +## Piper Voices + +Download one or more [Piper voices][piper voices] (both the `.onnx` and `.onnx.json` files for each voice). [Audio samples][piper samples] are available. + +As an example, we'll download the U.S. English "lessac" voice in medium quality: + +``` sh +mkdir -p voices +wget -O voices/en_US-lessac-medium.onnx 'https://huggingface.co/rhasspy/piper-voices/resolve/main/en/en_US/lessac/medium/en_US-lessac-medium.onnx?download=true' +wget -O voices/en_US-lessac-medium.onnx.json 'https://huggingface.co/rhasspy/piper-voices/resolve/main/en/en_US/lessac/medium/en_US-lessac-medium.onnx.json?download=true' +``` + +Generate a small set of samples with the CLI: + +``` sh +python3 generate_samples.py 'okay piper.' --model voices/en_US-lessac-medium.onnx --max-samples 10 --output-dir okay_piper/ +``` + +Check the `okay_piper/` directory for 10 WAV files (named `0.wav` to `9.wav`). + +You can add multiple `--model ` arguments to cycle between different voices when generating samples. + +See `--help` for more options, including `--length-scales` (speaking speeds). + +## Generator + +Download the LibriTTS-R generator (exported from [checkpoint][]): ``` sh wget -O models/en-us-libritts-high.pt 'https://github.com/rhasspy/piper-sample-generator/releases/download/v2.0.0/en_US-libritts_r-medium.pt' ``` -See links above for models for other languages. - -## Run - Generate a small set of samples with the CLI: ``` sh -python3 generate_samples.py 'okay, piper.' --max-samples 10 --output-dir okay_piper/ +python3 generate_samples.py 'okay piper.' --model models/en-us-libritts-high.pt --max-samples 10 --output-dir okay_piper/ ``` Check the `okay_piper/` directory for 10 WAV files (named `0.wav` to `9.wav`). @@ -45,63 +61,38 @@ Check the `okay_piper/` directory for 10 WAV files (named `0.wav` to `9.wav`). Generation can be much faster and more efficient if you have a GPU available and PyTorch is configured to use it. In this case, increase the batch size: ``` sh -python3 generate_samples.py 'okay, piper.' --max-samples 100 --batch-size 10 --output-dir okay_piper/ +python3 generate_samples.py 'okay piper.' --model models/en-us-libritts-high.pt --max-samples 100 --batch-size 10 --output-dir okay_piper/ ``` On an NVidia 2080 Ti with 11GB, a batch size of 100 was possible (generating approximately 100 samples per second). Setting `--max-speakers` to a value less than 904 (the number of speakers LibriTTS) is recommended. Because very few samples of later speakers were in the original dataset, using them can cause audio artifacts. -See `--help` for more options, including adjust the `--length-scales` (speaking speeds) and `--slerp-weights` (speaker blending) which are cycled per batch. - -Alternatively, you can import the generate function into another Python script: - -```python -from generate_samples import generate_samples # make sure to add this to your Python path as needed - -generate_samples(text = ["okay, piper"], max_samples = 100, output_dir = output_dir, batch_size=10) -``` - -There are some additional arguments available when importing the function directly, see the docstring of `generate_sample` for more information. +See `--help` for more options, including the `--length-scales` (speaking speeds) and `--slerp-weights` (speaker blending) which are cycled per batch. ### Augmentation -Once you have samples generating, you can augment them using [audiomentation](https://iver56.github.io/audiomentations/): +Once you have samples generated, you can augment them using [audiomentation](https://iver56.github.io/audiomentations/): ``` sh -python3 augment.py --sample-rate 16000 okay_piper/ okay_piper_augmented/ +python3 augment.py --sample-rate 22050 okay_piper/ okay_piper_augmented/ ``` This will do several things to each sample: 1. Randomly decrease the volume * The original samples are normalized, so different volume levels are needed -2. Randomly [apply an impulse response](https://iver56.github.io/audiomentations/waveform_transforms/apply_impulse_response/) using the files in `impulses/` +2. Randomly apply an [impulse response][] using the files in `impulses/` * Change the acoustics of the sample to sound like the speaker was in a room with echo or using a poor quality microphone -3. Resample to 16Khz for training (e.g., [openWakeWord](https://github.com/dscripka/openWakeWord)) +3. Resample to 16Khz for training (e.g., [openWakeWord][]) -## Short Phrases - -Models that were trained on audio books tend to perform poorly when speaking short phrases or single words. -The French, German, and Dutch models trained from the [MLS](http://openslr.org/94/) have this problem. - -The problem can be mitigated by repeating the phrase over and over, and then clipping out a single sample. -To do this automatically, follow these steps: - -1. Ensure your short phrase ends with a comma (`,`) -2. Lower the noise settings with `--noise-scales 0.333` and `--noise-scale-ws 0.333` -3. Use `--min-phoneme-count 300` (the value 300 was determined empirically and may be less for some models) - -For example: - -``` sh -python3 generate_samples.py \ - 'framboise,' \ - --model models/fr_FR-mls-medium.pt \ - --noise-scales 0.333 \ - --noise-scale-ws 0.333 \ - --min-phoneme-count 300 - --max-samples 1 \ - --output-dir . -``` + +[piper]: https://github.com/OHF-Voice/piper1-gpl/ +[openWakeWord]: https://github.com/dscripka/openWakeWord +[microWakeWord]: https://github.com/kahrendt/microWakeWord/ +[piper voices]: https://huggingface.co/rhasspy/piper-voices +[generator]: https://github.com/rhasspy/piper-sample-generator/releases/download/v2.0.0/en_US-libritts_r-medium.pt +[piper samples]: https://rhasspy.github.io/piper-samples/ +[checkpoint]: https://huggingface.co/datasets/rhasspy/piper-checkpoints/tree/main/en/en_US/libritts_r/medium +[impulse response]: https://iver56.github.io/audiomentations/waveform_transforms/apply_impulse_response/ diff --git a/augment.py b/augment.py index 834d223..269b9e9 100644 --- a/augment.py +++ b/augment.py @@ -22,7 +22,7 @@ def main() -> None: augment = Compose( transforms=[ - Gain(min_gain_in_db=-12, max_gain_in_db=0), + Gain(min_gain_db=-12, max_gain_db=0), ApplyImpulseResponse(impulses), ] ) diff --git a/generate_samples.py b/generate_samples.py index c12acaf..57c09c2 100755 --- a/generate_samples.py +++ b/generate_samples.py @@ -26,9 +26,9 @@ logging.basicConfig(level=logging.DEBUG) def generate_samples( text: Union[List[str], str], output_dir: Union[str, Path], + model: Union[str, Path], max_samples: Optional[int] = None, file_names: Optional[Iterable[str]] = None, - model: Union[str, Path] = _DIR / "models" / "en_US-libritts_r-medium.pt", batch_size: int = 1, slerp_weights: Tuple[float, ...] = (0.5,), length_scales: Tuple[float, ...] = (0.75, 1, 1.25), @@ -45,10 +45,10 @@ def generate_samples( text (List[str]): The text to convert into speech. Can be either a a list of strings, or a path to a file with text on each line. output_dir (str): The location to save the generated clips. + model (str): The path to the TTS generator model (.pt). max_samples (int): The maximum number of samples to generate. file_names (List[str]): The names to use when saving the files. Must be the same length as the `text` argument, if a list. - model (str): The path to the TTS generator model (.pt). batch_size (int): The batch size to use when generated the clips slerp_weights (List[float]): The weights to use when mixing speakers via SLERP. length_scales (List[float]): Controls the average duration/speed of the generated speech. @@ -100,17 +100,6 @@ def generate_samples( ) ) - # Define resampler to get to 16khz (https://pytorch.org/audio/stable/tutorials/audio_resampling_tutorial.html#kaiser-best) - resample_rate = 16000 - resampler = torchaudio.transforms.Resample( - sample_rate, - resample_rate, - lowpass_filter_width=64, - rolloff=0.9475937167399596, - # resampling_method="sinc_interp_kaiser", - beta=14.769656459379492, - ) - speakers_iter = it.cycle(it.product(range(num_speakers), range(num_speakers))) speakers_batch = list(it.islice(speakers_iter, 0, batch_size)) if isinstance(text, str) and os.path.exists(text): @@ -157,22 +146,23 @@ def generate_samples( return padded_lists phoneme_ids_by_batch = right_pad_lists(phoneme_ids_by_batch) - audio = generate_audio( - torch_model, - speaker_1, - speaker_2, - phoneme_ids_by_batch, - slerp_weight, - noise_scale, - noise_scale_w, - length_scale, - max_len, + audio = ( + generate_audio( + torch_model, + speaker_1, + speaker_2, + phoneme_ids_by_batch, + slerp_weight, + noise_scale, + noise_scale_w, + length_scale, + max_len, + ) + .cpu() + .numpy() ) - # Resample audio - audio_np = resampler(audio.cpu()).numpy() - - audio_int16 = audio_float_to_int16(audio_np) + audio_int16 = audio_float_to_int16(audio) for audio_idx in range(audio_int16.shape[0]): audio_data = np.trim_zeros(audio_int16[audio_idx].flatten()) @@ -183,7 +173,7 @@ def generate_samples( wav_file: wave.Wave_write = wave.open(str(wav_path), "wb") with wav_file: - wav_file.setframerate(resample_rate) + wav_file.setframerate(sample_rate) wav_file.setsampwidth(2) wav_file.setnchannels(1) wav_file.writeframes(audio_data) @@ -209,7 +199,7 @@ def generate_samples( def generate_samples_onnx( text: Union[List[str], str], output_dir: Union[str, Path], - model: Union[str, Path], + model: Union[str, Path, List[Union[str, Path]]], max_samples: Optional[int] = None, file_names: Optional[Iterable[str]] = None, length_scales: Tuple[float, ...] = (0.75, 1, 1.25), @@ -241,21 +231,20 @@ def generate_samples_onnx( if max_samples is None: max_samples = len(text) + if not isinstance(model, list): + model = [model] + _LOGGER.debug("Loading %s", model) - voice = PiperVoice.load(model, use_cuda=torch.cuda.is_available()) - _LOGGER.info("Successfully loaded the model") + voices = [PiperVoice.load(m, use_cuda=torch.cuda.is_available()) for m in model] + _LOGGER.info("Successfully loaded model(s)") output_dir = Path(output_dir) output_dir.mkdir(parents=True, exist_ok=True) - num_speakers = voice.config.num_speakers - if max_speakers is not None: - num_speakers = min(num_speakers, max_speakers) - sample_idx = 0 settings_iter = it.cycle( it.product( - list(range(num_speakers)), + voices, length_scales, noise_scales, noise_scale_ws, @@ -278,27 +267,32 @@ def generate_samples_onnx( if file_names: file_names = it.cycle(file_names) - for speaker_id, length_scale, noise_scale, noise_w_scale in settings_iter: - if isinstance(file_names, it.cycle): - wav_path = output_dir / next(file_names) - else: - wav_path = output_dir / f"{sample_idx}.wav" + for voice, length_scale, noise_scale, noise_w_scale in settings_iter: + num_speakers = voice.config.num_speakers + if max_speakers is not None: + num_speakers = min(num_speakers, max_speakers) - wav_file: wave.Wave_write = wave.open(str(wav_path), "wb") - voice.synthesize_wav( - next(texts), - wav_file=wav_file, - syn_config=SynthesisConfig( - speaker_id=speaker_id, - length_scale=length_scale, - noise_scale=noise_scale, - noise_w_scale=noise_w_scale, - ), - ) + for speaker_id in range(num_speakers): + if isinstance(file_names, it.cycle): + wav_path = output_dir / next(file_names) + else: + wav_path = output_dir / f"{sample_idx}.wav" - sample_idx += 1 - if sample_idx >= max_samples: - break + wav_file: wave.Wave_write = wave.open(str(wav_path), "wb") + voice.synthesize_wav( + next(texts), + wav_file=wav_file, + syn_config=SynthesisConfig( + speaker_id=speaker_id, + length_scale=length_scale, + noise_scale=noise_scale, + noise_w_scale=noise_w_scale, + ), + ) + + sample_idx += 1 + if sample_idx >= max_samples: + return _LOGGER.info("Done") @@ -459,45 +453,92 @@ def audio_float_to_int16( # ----------------------------------------------------------------------------- -def main() -> None: +def main() -> int: """Main entry point.""" # Get command line arguments parser = argparse.ArgumentParser() parser.add_argument("text") - parser.add_argument("--max-samples", required=True, type=int) parser.add_argument( - "--model", default=_DIR / "models" / "en_US-libritts_r-medium.pt" + "--max-samples", + required=True, + type=int, + help="Maximum number of samples to generate", ) - parser.add_argument("--batch-size", type=int, default=1) - parser.add_argument("--slerp-weights", nargs="+", type=float, default=[0.5]) parser.add_argument( - "--length-scales", nargs="+", type=float, default=[1.0, 0.75, 1.25, 1.4] + "--model", + required=True, + action="append", + help="Path to PyTorch generator (.pt) or Piper voice model (.onnx)", + ) + parser.add_argument( + "--batch-size", type=int, default=1, help="CUDA batch size (generator only)" + ) + parser.add_argument( + "--slerp-weights", + nargs="+", + type=float, + default=[0.5], + help="Speaker blending weights (generator only)", + ) + parser.add_argument( + "--length-scales", + nargs="+", + type=float, + default=[1.0, 0.75, 1.25, 1.4], + help="Audio length scales (< 1 is faster, > 1 is slower)", ) parser.add_argument( "--noise-scales", nargs="+", type=float, default=[0.667, 0.75, 0.85, 0.9, 1.0, 1.4], + help="Noise amounts added to audio (most voices use 0.667)", + ) + parser.add_argument( + "--noise-scale-ws", + nargs="+", + type=float, + default=[0.8], + help="Phoneme width variation (most voices use 0.8)", + ) + parser.add_argument( + "--output-dir", + default="output", + help="Directory to output WAV files (default: ./output)", ) - parser.add_argument("--noise-scale-ws", nargs="+", type=float, default=[0.8]) - parser.add_argument("--output-dir", default="output") parser.add_argument( "--max-speakers", type=int, - help="Maximum number of speakers to use (default: all)", + help="Maximum number of speakers to use (default: no limit)", ) parser.add_argument("--verbose", action="store_true") args = parser.parse_args().__dict__ # Generate speech - model_path = Path(args["model"]) - if model_path.suffix == ".onnx": + model_paths = [Path(m) for m in args["model"]] + assert model_paths + + if any(mp for mp in model_paths[1:] if mp.suffix != model_paths[0].suffix): + _LOGGER.error("All models must have the same suffix (.pt or .onnx)") + return 1 + + if model_paths[0].suffix == ".onnx": # Use Piper voice (.onnx) generate_samples_onnx(**args) - else: + elif model_paths[0].suffix == ".pt": # Use PyTorch generator (.pt) + if len(model_paths) > 1: + _LOGGER.error("Only one generator (.pt) is supported") + return 1 + + args["model"] = args["model"][0] generate_samples(**args) + else: + _LOGGER.error("Models must have .pt or .onnx suffix") + return 1 + + return 0 if __name__ == "__main__": From 5e67370ab3fa4be23f6d856e085d5ab9ed5d46ea Mon Sep 17 00:00:00 2001 From: Michael Hansen Date: Fri, 29 Aug 2025 15:17:12 -0500 Subject: [PATCH 08/16] Update changelog --- CHANGELOG.md | 1 + 1 file changed, 1 insertion(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index d3bb09a..39225c1 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -5,5 +5,6 @@ - Move phonemization to piper 1.3.0 (piper-phonemize is deprecated) - Move to PyTorch 2 - Add support for using Piper voices (`.onnx`) directly +- Allow multiple `--model` for Piper voices (`.onnx`) - Remove silence trimming - Remove `min-phoneme-count` From af7ac7aae2bd2db01caf657d7cf60d2622008a04 Mon Sep 17 00:00:00 2001 From: Kevin Ahrendt Date: Fri, 19 Sep 2025 13:18:06 -0400 Subject: [PATCH 09/16] add support for MPS acceleration --- generate_samples.py | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) diff --git a/generate_samples.py b/generate_samples.py index 57c09c2..edf4e12 100755 --- a/generate_samples.py +++ b/generate_samples.py @@ -1,5 +1,6 @@ #!/usr/bin/env python3 import argparse +import gc import itertools as it import json import logging @@ -73,6 +74,10 @@ def generate_samples( if torch.cuda.is_available(): torch_model.cuda() _LOGGER.debug("CUDA available, using GPU") + elif torch.backends.mps.is_available(): + mps_device = torch.device("mps") + torch_model.to(mps_device) + _LOGGER.debug("MPS available, using GPU") output_dir = Path(output_dir) output_dir.mkdir(parents=True, exist_ok=True) @@ -162,6 +167,11 @@ def generate_samples( .numpy() ) + if torch.backends.mps.is_available(): + # There seems to be a memory leak if we don't empty the cache after each batch with mps + torch.mps.empty_cache() + gc.collect() + audio_int16 = audio_float_to_int16(audio) for audio_idx in range(audio_int16.shape[0]): audio_data = np.trim_zeros(audio_int16[audio_idx].flatten()) @@ -319,6 +329,12 @@ def generate_audio( speaker_2 = speaker_2.cuda() x = cast(torch.LongTensor, x.cuda()) x_lengths = cast(torch.LongTensor, x_lengths.cuda()) + elif torch.backends.mps.is_available(): + mps_device = torch.device("mps") + speaker_1 = speaker_1.to(mps_device) + speaker_2 = speaker_2.to(mps_device) + x = x.to(mps_device) + x_lengths = x_lengths.to(mps_device) x, m_p_orig, logs_p_orig, x_mask = model.enc_p(x, x_lengths) emb0 = model.emb_g(speaker_1) From df5799c141e0603fe17bd6a6f96661ed3aa13f1a Mon Sep 17 00:00:00 2001 From: Kevin Ahrendt Date: Fri, 19 Sep 2025 14:42:49 -0400 Subject: [PATCH 10/16] add support for directly inputting phonemes --- generate_samples.py | 116 +++++++++++++++++++++++++++++++++----------- 1 file changed, 87 insertions(+), 29 deletions(-) diff --git a/generate_samples.py b/generate_samples.py index 57c09c2..5af999e 100755 --- a/generate_samples.py +++ b/generate_samples.py @@ -36,6 +36,7 @@ def generate_samples( noise_scale_ws: Tuple[float, ...] = (0.8,), max_speakers: Optional[int] = None, verbose: bool = False, + phoneme_input: bool = False, **kwargs, ) -> None: """ @@ -56,6 +57,7 @@ def generate_samples( noise_scale_ws (List[float]): A parameter for the stochastic duration of words/phonemes. max_speakers (int): The maximum speaker number to use, if the model is multi-speaker. verbose (bool): Enable or disable more detailed logging messages (default: False). + phoneme_input (bool): Set to indicate given input text is phoneme input. Returns: None """ @@ -132,7 +134,7 @@ def generate_samples( phoneme_ids_by_batch = [] for i in range(batch_size): - phoneme_ids = get_phonemes(voice, config, next(texts), verbose) + phoneme_ids = get_phonemes(voice, config, next(texts), verbose, phoneme_input) phoneme_ids_by_batch.append(phoneme_ids) def right_pad_lists(lists): @@ -146,22 +148,27 @@ def generate_samples( return padded_lists phoneme_ids_by_batch = right_pad_lists(phoneme_ids_by_batch) - audio = ( - generate_audio( - torch_model, - speaker_1, - speaker_2, - phoneme_ids_by_batch, - slerp_weight, - noise_scale, - noise_scale_w, - length_scale, - max_len, - ) - .cpu() - .numpy() + audio, phoneme_samples = generate_audio( + torch_model, + speaker_1, + speaker_2, + phoneme_ids_by_batch, + slerp_weight, + noise_scale, + noise_scale_w, + length_scale, + max_len, ) + # Trim audio to actual length based on phoneme samples + for i in range(audio.shape[0]): + # Fill time after last speech with silence (zeros) + # It will be removed in the next stage with np.trim_zeros + last_sample_idx = int(phoneme_samples[i].flatten().sum().item()) + audio[i, 0, last_sample_idx+1:] = 0 + + audio = audio.cpu().numpy() + audio_int16 = audio_float_to_int16(audio) for audio_idx in range(audio_int16.shape[0]): audio_data = np.trim_zeros(audio_int16[audio_idx].flatten()) @@ -206,6 +213,7 @@ def generate_samples_onnx( noise_scales: Tuple[float, ...] = (0.667,), noise_scale_ws: Tuple[float, ...] = (0.8,), max_speakers: Optional[int] = None, + phoneme_input: bool = False, **kwargs, ) -> None: """ @@ -223,6 +231,7 @@ def generate_samples_onnx( noise_scales (List[float]): A parameter for overall variability of the generated speech. noise_scale_ws (List[float]): A parameter for the stochastic duration of words/phonemes. max_speakers (int): The maximum speaker number to use, if the model is multi-speaker. + phoneme_input (bool): Set to indicate given input text is phoneme input. Returns: None @@ -278,17 +287,60 @@ def generate_samples_onnx( else: wav_path = output_dir / f"{sample_idx}.wav" - wav_file: wave.Wave_write = wave.open(str(wav_path), "wb") - voice.synthesize_wav( - next(texts), - wav_file=wav_file, - syn_config=SynthesisConfig( + text_input = next(texts) + + if phoneme_input: + # For ONNX models with phoneme input, build phoneme IDs manually + phonemes = [p for p in list(text_input)] + + # Build phoneme IDs similar to get_phonemes function + id_map = voice.config.phoneme_id_map + + # Beginning of utterance + phoneme_ids = list(id_map.get("^", [1])) # Default to [1] if not found + phoneme_ids.extend(id_map.get("_", [0])) # Default to [0] if not found + + # Add phonemes + for phoneme in phonemes: + p_ids = id_map.get(phoneme) + if p_ids is not None: + phoneme_ids.extend(p_ids) + phoneme_ids.extend(id_map.get("_", [0])) + else: + _LOGGER.debug(f"Phoneme '{phoneme}' not found in model's phoneme map") + + # End of utterance + phoneme_ids.extend(id_map.get("$", [2])) # Default to [2] if not found + + # Generate audio from phoneme IDs + syn_config = SynthesisConfig( speaker_id=speaker_id, length_scale=length_scale, noise_scale=noise_scale, noise_w_scale=noise_w_scale, - ), - ) + ) + audio = voice.phoneme_ids_to_audio(phoneme_ids, syn_config) + + # Convert to int16 and write to WAV + audio_int16 = audio_float_to_int16(audio[np.newaxis, :]) + wav_file: wave.Wave_write = wave.open(str(wav_path), "wb") + with wav_file: + wav_file.setframerate(voice.config.sample_rate) + wav_file.setsampwidth(2) + wav_file.setnchannels(1) + wav_file.writeframes(audio_int16.flatten()) + else: + wav_file: wave.Wave_write = wave.open(str(wav_path), "wb") + voice.synthesize_wav( + text_input, + wav_file=wav_file, + syn_config=SynthesisConfig( + speaker_id=speaker_id, + length_scale=length_scale, + noise_scale=noise_scale, + noise_w_scale=noise_w_scale, + ), + ) sample_idx += 1 if sample_idx >= max_samples: @@ -310,7 +362,7 @@ def generate_audio( noise_scale_w, length_scale, max_len, -) -> torch.FloatTensor: +) -> Tuple[torch.FloatTensor, torch.FloatTensor]: x = torch.LongTensor(phoneme_ids) x_lengths = torch.LongTensor([len(i) for i in phoneme_ids]) @@ -350,8 +402,9 @@ def generate_audio( o = model.dec((z * y_mask)[:, :, :max_len], g=g) audio = o + phoneme_samples = w_ceil * 256 # hop length - return audio + return audio, phoneme_samples _PHONEMIZER = EspeakPhonemizer() @@ -362,13 +415,17 @@ def get_phonemes( config: Dict[str, Any], text: str, verbose: bool = False, + phoneme_input: bool = False, ) -> List[int]: # Combine all sentences - phonemes = [ - p - for sentence_phonemes in _PHONEMIZER.phonemize(voice, text) - for p in sentence_phonemes - ] + if phoneme_input: + phonemes = [p for p in list(text)] + else: + phonemes = [ + p + for sentence_phonemes in _PHONEMIZER.phonemize(voice, text) + for p in sentence_phonemes + ] if verbose is True: _LOGGER.debug("Phonemes: %s", phonemes) @@ -512,6 +569,7 @@ def main() -> int: type=int, help="Maximum number of speakers to use (default: no limit)", ) + parser.add_argument("--phoneme-input", action="store_true", help="Treat input text as phoneme input") parser.add_argument("--verbose", action="store_true") args = parser.parse_args().__dict__ From 66ec23fa497823a75577d25c2ed6d77a04db9413 Mon Sep 17 00:00:00 2001 From: Michael Hansen Date: Fri, 19 Sep 2025 16:21:32 -0500 Subject: [PATCH 11/16] Clean up --- CHANGELOG.md | 5 ++++ generate_samples.py | 57 ++++++++++++++++++++++++--------------------- pyproject.toml | 2 +- 3 files changed, 37 insertions(+), 27 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 39225c1..9d6cbf8 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,10 @@ # Changelog +## 3.1.0 + +- Support MPS acceleration on Apple Silicon +- Add `--phoneme-input` flag + ## 3.0.0 - Move phonemization to piper 1.3.0 (piper-phonemize is deprecated) diff --git a/generate_samples.py b/generate_samples.py index 60ce657..ee95845 100755 --- a/generate_samples.py +++ b/generate_samples.py @@ -5,6 +5,7 @@ import itertools as it import json import logging import os +import unicodedata import wave from collections.abc import Iterable from pathlib import Path @@ -12,13 +13,11 @@ from typing import Any, Dict, List, Optional, Tuple, Union, cast import numpy as np import torch -import torchaudio from piper import PiperVoice, SynthesisConfig from piper.phonemize_espeak import EspeakPhonemizer from piper_train.vits import commons -_DIR = Path(__file__).parent _LOGGER = logging.getLogger(__name__) logging.basicConfig(level=logging.DEBUG) @@ -139,7 +138,9 @@ def generate_samples( phoneme_ids_by_batch = [] for i in range(batch_size): - phoneme_ids = get_phonemes(voice, config, next(texts), verbose, phoneme_input) + phoneme_ids = get_phonemes( + voice, config, next(texts), verbose, phoneme_input + ) phoneme_ids_by_batch.append(phoneme_ids) def right_pad_lists(lists): @@ -170,16 +171,16 @@ def generate_samples( # Fill time after last speech with silence (zeros) # It will be removed in the next stage with np.trim_zeros last_sample_idx = int(phoneme_samples[i].flatten().sum().item()) - audio[i, 0, last_sample_idx+1:] = 0 + audio[i, 0, last_sample_idx + 1 :] = 0 - audio = audio.cpu().numpy() + audio_numpy = audio.cpu().numpy() if torch.backends.mps.is_available(): # There seems to be a memory leak if we don't empty the cache after each batch with mps torch.mps.empty_cache() gc.collect() - audio_int16 = audio_float_to_int16(audio) + audio_int16 = audio_float_to_int16(audio_numpy) for audio_idx in range(audio_int16.shape[0]): audio_data = np.trim_zeros(audio_int16[audio_idx].flatten()) @@ -301,14 +302,14 @@ def generate_samples_onnx( if phoneme_input: # For ONNX models with phoneme input, build phoneme IDs manually - phonemes = [p for p in list(text_input)] + phonemes = list(unicodedata.normalize("NFD", text_input)) # Build phoneme IDs similar to get_phonemes function id_map = voice.config.phoneme_id_map # Beginning of utterance phoneme_ids = list(id_map.get("^", [1])) # Default to [1] if not found - phoneme_ids.extend(id_map.get("_", [0])) # Default to [0] if not found + phoneme_ids.extend(id_map.get("_", [0])) # Default to [0] if not found # Add phonemes for phoneme in phonemes: @@ -317,7 +318,9 @@ def generate_samples_onnx( phoneme_ids.extend(p_ids) phoneme_ids.extend(id_map.get("_", [0])) else: - _LOGGER.debug(f"Phoneme '{phoneme}' not found in model's phoneme map") + _LOGGER.warning( + "Phoneme '%s' not found in model's phoneme map", phoneme + ) # End of utterance phoneme_ids.extend(id_map.get("$", [2])) # Default to [2] if not found @@ -340,17 +343,17 @@ def generate_samples_onnx( wav_file.setnchannels(1) wav_file.writeframes(audio_int16.flatten()) else: - wav_file: wave.Wave_write = wave.open(str(wav_path), "wb") - voice.synthesize_wav( - text_input, - wav_file=wav_file, - syn_config=SynthesisConfig( - speaker_id=speaker_id, - length_scale=length_scale, - noise_scale=noise_scale, - noise_w_scale=noise_w_scale, - ), - ) + with wave.open(str(wav_path), "wb") as wav_file: + voice.synthesize_wav( + text_input, + wav_file=wav_file, + syn_config=SynthesisConfig( + speaker_id=speaker_id, + length_scale=length_scale, + noise_scale=noise_scale, + noise_w_scale=noise_w_scale, + ), + ) sample_idx += 1 if sample_idx >= max_samples: @@ -385,8 +388,8 @@ def generate_audio( mps_device = torch.device("mps") speaker_1 = speaker_1.to(mps_device) speaker_2 = speaker_2.to(mps_device) - x = x.to(mps_device) - x_lengths = x_lengths.to(mps_device) + x = cast(torch.LongTensor, x.to(mps_device)) + x_lengths = cast(torch.LongTensor, x_lengths.to(mps_device)) x, m_p_orig, logs_p_orig, x_mask = model.enc_p(x, x_lengths) emb0 = model.emb_g(speaker_1) @@ -417,8 +420,8 @@ def generate_audio( z = model.flow(z_p, y_mask, g=g, reverse=True) o = model.dec((z * y_mask)[:, :, :max_len], g=g) - audio = o - phoneme_samples = w_ceil * 256 # hop length + audio = cast(torch.FloatTensor, o) + phoneme_samples = cast(torch.FloatTensor, w_ceil * 256) # hop length return audio, phoneme_samples @@ -435,7 +438,7 @@ def get_phonemes( ) -> List[int]: # Combine all sentences if phoneme_input: - phonemes = [p for p in list(text)] + phonemes = list(unicodedata.normalize("NFD", text)) else: phonemes = [ p @@ -585,7 +588,9 @@ def main() -> int: type=int, help="Maximum number of speakers to use (default: no limit)", ) - parser.add_argument("--phoneme-input", action="store_true", help="Treat input text as phoneme input") + parser.add_argument( + "--phoneme-input", action="store_true", help="Treat input text as phoneme input" + ) parser.add_argument("--verbose", action="store_true") args = parser.parse_args().__dict__ diff --git a/pyproject.toml b/pyproject.toml index 5061a05..36c9711 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta" [project] name = "piper-sample-generator" -version = "3.0.0" +version = "3.1.0" license = {text = "Apache-2.0"} description = "Generate TTS audio samples for training wake word systems" readme = "README.md" From ded9350eaff558af07f312464ac71baf7de834df Mon Sep 17 00:00:00 2001 From: Michael Hansen Date: Fri, 14 Nov 2025 10:47:29 -0600 Subject: [PATCH 12/16] Check if file --- generate_samples.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/generate_samples.py b/generate_samples.py index ee95845..753e60e 100755 --- a/generate_samples.py +++ b/generate_samples.py @@ -108,7 +108,7 @@ def generate_samples( speakers_iter = it.cycle(it.product(range(num_speakers), range(num_speakers))) speakers_batch = list(it.islice(speakers_iter, 0, batch_size)) - if isinstance(text, str) and os.path.exists(text): + if isinstance(text, str) and os.path.isfile(text): texts = it.cycle( [ i.strip() From c9d824c0e2cce8bdeb000c219dc9cbc84df086ea Mon Sep 17 00:00:00 2001 From: Michael Hansen Date: Thu, 12 Mar 2026 10:48:04 -0500 Subject: [PATCH 13/16] Require sample rate in augment --- augment.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/augment.py b/augment.py index 269b9e9..dcc16cd 100644 --- a/augment.py +++ b/augment.py @@ -15,7 +15,7 @@ def main() -> None: parser = argparse.ArgumentParser() parser.add_argument("input_dir") parser.add_argument("output_dir") - parser.add_argument("--sample-rate", type=int) + parser.add_argument("--sample-rate", type=int, required=True) args = parser.parse_args() impulses = list((_DIR / "impulses").glob("*.wav")) From 1a8c49bd29b3a132721086ee88f2253f788594a8 Mon Sep 17 00:00:00 2001 From: Michael Hansen Date: Thu, 12 Mar 2026 15:16:51 -0500 Subject: [PATCH 14/16] Move to package --- CHANGELOG.md | 4 +++ README.md | 10 +++--- piper_sample_generator/__init__.py | 1 + .../__main__.py | 8 +++-- .../augment.py | 10 +++--- .../impulses}/Accoustic2_Impulse.wav | Bin .../impulses}/Blatty Plate.wav | Bin .../impulses}/Concrete Room.wav | Bin .../impulses}/Derlon Sanctuary.wav | Bin .../impulses}/Fat Bass.wav | Bin .../impulses}/Reverse Gate.wav | Bin .../impulses}/Symphonic.wav | Bin .../impulses}/ir_bathroom1.wav | Bin pylintrc | 12 +++---- pyproject.toml | 34 +++++++----------- requirements.txt | 6 ---- requirements_dev.txt | 5 --- script/format | 6 ++-- script/lint | 12 +++---- script/run | 2 +- 20 files changed, 47 insertions(+), 63 deletions(-) create mode 100644 piper_sample_generator/__init__.py rename generate_samples.py => piper_sample_generator/__main__.py (99%) rename augment.py => piper_sample_generator/augment.py (91%) rename {impulses => piper_sample_generator/impulses}/Accoustic2_Impulse.wav (100%) rename {impulses => piper_sample_generator/impulses}/Blatty Plate.wav (100%) rename {impulses => piper_sample_generator/impulses}/Concrete Room.wav (100%) rename {impulses => piper_sample_generator/impulses}/Derlon Sanctuary.wav (100%) rename {impulses => piper_sample_generator/impulses}/Fat Bass.wav (100%) rename {impulses => piper_sample_generator/impulses}/Reverse Gate.wav (100%) rename {impulses => piper_sample_generator/impulses}/Symphonic.wav (100%) rename {impulses => piper_sample_generator/impulses}/ir_bathroom1.wav (100%) delete mode 100644 requirements.txt delete mode 100644 requirements_dev.txt diff --git a/CHANGELOG.md b/CHANGELOG.md index 9d6cbf8..916341d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,9 @@ # Changelog +## 3.2.0 + +- Refactor as `piper_sample_generator` package + ## 3.1.0 - Support MPS acceleration on Apple Silicon diff --git a/README.md b/README.md index fdc4117..4b4fd07 100644 --- a/README.md +++ b/README.md @@ -33,7 +33,7 @@ wget -O voices/en_US-lessac-medium.onnx.json 'https://huggingface.co/rhasspy/pip Generate a small set of samples with the CLI: ``` sh -python3 generate_samples.py 'okay piper.' --model voices/en_US-lessac-medium.onnx --max-samples 10 --output-dir okay_piper/ +python3 -m piper_sample_generator 'okay piper.' --model voices/en_US-lessac-medium.onnx --max-samples 10 --output-dir okay_piper/ ``` Check the `okay_piper/` directory for 10 WAV files (named `0.wav` to `9.wav`). @@ -53,7 +53,7 @@ wget -O models/en-us-libritts-high.pt 'https://github.com/rhasspy/piper-sample-g Generate a small set of samples with the CLI: ``` sh -python3 generate_samples.py 'okay piper.' --model models/en-us-libritts-high.pt --max-samples 10 --output-dir okay_piper/ +python3 -m piper_sample_generator 'okay piper.' --model models/en-us-libritts-high.pt --max-samples 10 --output-dir okay_piper/ ``` Check the `okay_piper/` directory for 10 WAV files (named `0.wav` to `9.wav`). @@ -61,7 +61,7 @@ Check the `okay_piper/` directory for 10 WAV files (named `0.wav` to `9.wav`). Generation can be much faster and more efficient if you have a GPU available and PyTorch is configured to use it. In this case, increase the batch size: ``` sh -python3 generate_samples.py 'okay piper.' --model models/en-us-libritts-high.pt --max-samples 100 --batch-size 10 --output-dir okay_piper/ +python3 -m piper_sample_generator 'okay piper.' --model models/en-us-libritts-high.pt --max-samples 100 --batch-size 10 --output-dir okay_piper/ ``` On an NVidia 2080 Ti with 11GB, a batch size of 100 was possible (generating approximately 100 samples per second). @@ -75,14 +75,14 @@ See `--help` for more options, including the `--length-scales` (speaking speeds) Once you have samples generated, you can augment them using [audiomentation](https://iver56.github.io/audiomentations/): ``` sh -python3 augment.py --sample-rate 22050 okay_piper/ okay_piper_augmented/ +python3 -m piper_sample_generator.augment --sample-rate 22050 okay_piper/ okay_piper_augmented/ ``` This will do several things to each sample: 1. Randomly decrease the volume * The original samples are normalized, so different volume levels are needed -2. Randomly apply an [impulse response][] using the files in `impulses/` +2. Randomly apply an [impulse response][] using the files in `piper_sample_generator/impulses/` * Change the acoustics of the sample to sound like the speaker was in a room with echo or using a poor quality microphone 3. Resample to 16Khz for training (e.g., [openWakeWord][]) diff --git a/piper_sample_generator/__init__.py b/piper_sample_generator/__init__.py new file mode 100644 index 0000000..0142775 --- /dev/null +++ b/piper_sample_generator/__init__.py @@ -0,0 +1 @@ +"""Piper sample generator.""" diff --git a/generate_samples.py b/piper_sample_generator/__main__.py similarity index 99% rename from generate_samples.py rename to piper_sample_generator/__main__.py index 753e60e..3074fb5 100755 --- a/generate_samples.py +++ b/piper_sample_generator/__main__.py @@ -16,7 +16,10 @@ import torch from piper import PiperVoice, SynthesisConfig from piper.phonemize_espeak import EspeakPhonemizer -from piper_train.vits import commons +try: + from piper_train.vits import commons +except ImportError: + from piper_train.vits import commons _LOGGER = logging.getLogger(__name__) logging.basicConfig(level=logging.DEBUG) @@ -176,7 +179,8 @@ def generate_samples( audio_numpy = audio.cpu().numpy() if torch.backends.mps.is_available(): - # There seems to be a memory leak if we don't empty the cache after each batch with mps + # There seems to be a memory leak if we don't empty the cache + # after each batch with mps torch.mps.empty_cache() gc.collect() diff --git a/augment.py b/piper_sample_generator/augment.py similarity index 91% rename from augment.py rename to piper_sample_generator/augment.py index dcc16cd..47a226d 100644 --- a/augment.py +++ b/piper_sample_generator/augment.py @@ -1,12 +1,11 @@ #!/usr/bin/env python3 import argparse import audioop -import sys import wave from pathlib import Path import numpy as np -from audiomentations import Compose, ApplyImpulseResponse, Gain +from audiomentations import ApplyImpulseResponse, Compose, Gain _DIR = Path(__file__).parent @@ -35,9 +34,10 @@ def main() -> None: output_wav = output_dir / (input_wav.relative_to(input_dir)) output_wav.parent.mkdir(parents=True, exist_ok=True) - with wave.open(str(input_wav), "rb") as input_wav_file, wave.open( - str(output_wav), "wb" - ) as output_wav_file: + with ( + wave.open(str(input_wav), "rb") as input_wav_file, + wave.open(str(output_wav), "wb") as output_wav_file, + ): assert input_wav_file.getsampwidth() == 2 assert input_wav_file.getnchannels() == 1 diff --git a/impulses/Accoustic2_Impulse.wav b/piper_sample_generator/impulses/Accoustic2_Impulse.wav similarity index 100% rename from impulses/Accoustic2_Impulse.wav rename to piper_sample_generator/impulses/Accoustic2_Impulse.wav diff --git a/impulses/Blatty Plate.wav b/piper_sample_generator/impulses/Blatty Plate.wav similarity index 100% rename from impulses/Blatty Plate.wav rename to piper_sample_generator/impulses/Blatty Plate.wav diff --git a/impulses/Concrete Room.wav b/piper_sample_generator/impulses/Concrete Room.wav similarity index 100% rename from impulses/Concrete Room.wav rename to piper_sample_generator/impulses/Concrete Room.wav diff --git a/impulses/Derlon Sanctuary.wav b/piper_sample_generator/impulses/Derlon Sanctuary.wav similarity index 100% rename from impulses/Derlon Sanctuary.wav rename to piper_sample_generator/impulses/Derlon Sanctuary.wav diff --git a/impulses/Fat Bass.wav b/piper_sample_generator/impulses/Fat Bass.wav similarity index 100% rename from impulses/Fat Bass.wav rename to piper_sample_generator/impulses/Fat Bass.wav diff --git a/impulses/Reverse Gate.wav b/piper_sample_generator/impulses/Reverse Gate.wav similarity index 100% rename from impulses/Reverse Gate.wav rename to piper_sample_generator/impulses/Reverse Gate.wav diff --git a/impulses/Symphonic.wav b/piper_sample_generator/impulses/Symphonic.wav similarity index 100% rename from impulses/Symphonic.wav rename to piper_sample_generator/impulses/Symphonic.wav diff --git a/impulses/ir_bathroom1.wav b/piper_sample_generator/impulses/ir_bathroom1.wav similarity index 100% rename from impulses/ir_bathroom1.wav rename to piper_sample_generator/impulses/ir_bathroom1.wav diff --git a/pylintrc b/pylintrc index 561b2f1..60fdb1d 100644 --- a/pylintrc +++ b/pylintrc @@ -1,3 +1,6 @@ +[MASTER] +ignored-modules=torch + [MESSAGES CONTROL] disable= format, @@ -31,14 +34,7 @@ disable= missing-class-docstring, missing-function-docstring, import-error, - consider-using-with + relative-beyond-top-level [FORMAT] expected-line-ending-format=LF - -[TYPECHECK] - -# List of members which are set dynamically and missed by pylint inference -# system, and so shouldn't trigger E1101 when accessed. Python regular -# expressions are accepted. -generated-members=numpy.*,torch.* diff --git a/pyproject.toml b/pyproject.toml index 36c9711..99a0bd1 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,41 +4,31 @@ build-backend = "setuptools.build_meta" [project] name = "piper-sample-generator" -version = "3.1.0" -license = {text = "Apache-2.0"} +version = "3.2.0" +license = {text = "MIT"} description = "Generate TTS audio samples for training wake word systems" readme = "README.md" authors = [ {name = "The Home Assistant Authors", email = "hello@home-assistant.io"} ] keywords = ["piper", "sample", "tts", "wakeword"] -classifiers = [ - "Development Status :: 3 - Alpha", - "Intended Audience :: Developers", - "Topic :: Text Processing :: Linguistic", - "License :: OSI Approved :: Apache Software License", - "Programming Language :: Python :: 3.9", - "Programming Language :: Python :: 3.10", - "Programming Language :: Python :: 3.11", - "Programming Language :: Python :: 3.12", - "Programming Language :: Python :: 3.13", -] requires-python = ">=3.9.0" dependencies = [ - "piper-tts>=1.3.0,<2", + "audiomentations==0.33.0", + "piper-tts==1.3.0", + "numpy>=2,<3", "torch>=2,<3", "torchaudio", - "audiomentations", - "numpy", + "webrtcvad", ] [project.optional-dependencies] dev = [ - "black==24.8.0", - "flake8==7.2.0", - "mypy==1.14.0", - "pylint==3.2.7", - "pytest==8.3.5", + "black==22.12.0", + "flake8==6.0.0", + "isort==5.11.3", + "mypy==0.991", + "pylint==2.15.9", ] [project.urls] @@ -49,4 +39,4 @@ platforms = ["any"] zip-safe = true [tool.setuptools.packages.find] -include = [] +include = ["piper_sample_generator*"] diff --git a/requirements.txt b/requirements.txt deleted file mode 100644 index 8443d84..0000000 --- a/requirements.txt +++ /dev/null @@ -1,6 +0,0 @@ -audiomentations==0.33.0 -piper-phonemize==1.1.0 -numpy<2 -torch<2 -torchaudio -webrtcvad diff --git a/requirements_dev.txt b/requirements_dev.txt deleted file mode 100644 index 77190e6..0000000 --- a/requirements_dev.txt +++ /dev/null @@ -1,5 +0,0 @@ -black==22.12.0 -flake8==6.0.0 -isort==5.11.3 -mypy==0.991 -pylint==2.15.9 diff --git a/script/format b/script/format index 7f04417..b8b283b 100755 --- a/script/format +++ b/script/format @@ -6,7 +6,7 @@ from pathlib import Path _DIR = Path(__file__).parent _PROGRAM_DIR = _DIR.parent _VENV_DIR = _PROGRAM_DIR / ".venv" -_SCRIPT = _PROGRAM_DIR / "generate_samples.py" +_MODULE_DIR = _PROGRAM_DIR / "piper_sample_generator" if _VENV_DIR.exists(): context = venv.EnvBuilder().ensure_directories(_VENV_DIR) @@ -14,5 +14,5 @@ if _VENV_DIR.exists(): else: python_exe = "python3" -subprocess.check_call([python_exe, "-m", "black", str(_SCRIPT)]) -subprocess.check_call([python_exe, "-m", "isort", str(_SCRIPT)]) +subprocess.check_call([python_exe, "-m", "black", str(_MODULE_DIR)]) +subprocess.check_call([python_exe, "-m", "isort", str(_MODULE_DIR)]) diff --git a/script/lint b/script/lint index e4231e0..34222f0 100755 --- a/script/lint +++ b/script/lint @@ -6,7 +6,7 @@ from pathlib import Path _DIR = Path(__file__).parent _PROGRAM_DIR = _DIR.parent _VENV_DIR = _PROGRAM_DIR / ".venv" -_SCRIPT = _PROGRAM_DIR / "generate_samples.py" +_MODULE_DIR = _PROGRAM_DIR / "piper_sample_generator" if _VENV_DIR.exists(): context = venv.EnvBuilder().ensure_directories(_VENV_DIR) @@ -14,8 +14,8 @@ if _VENV_DIR.exists(): else: python_exe = "python3" -subprocess.check_call([python_exe, "-m", "black", str(_SCRIPT), "--check"]) -subprocess.check_call([python_exe, "-m", "isort", str(_SCRIPT), "--check"]) -subprocess.check_call([python_exe, "-m", "flake8", str(_SCRIPT)]) -subprocess.check_call([python_exe, "-m", "pylint", str(_SCRIPT)]) -subprocess.check_call([python_exe, "-m", "mypy", str(_SCRIPT)]) +subprocess.check_call([python_exe, "-m", "black", str(_MODULE_DIR), "--check"]) +subprocess.check_call([python_exe, "-m", "isort", str(_MODULE_DIR), "--check"]) +subprocess.check_call([python_exe, "-m", "flake8", str(_MODULE_DIR)]) +subprocess.check_call([python_exe, "-m", "pylint", str(_MODULE_DIR)]) +subprocess.check_call([python_exe, "-m", "mypy", str(_MODULE_DIR)]) diff --git a/script/run b/script/run index f2837d0..ae921c9 100755 --- a/script/run +++ b/script/run @@ -14,4 +14,4 @@ if _VENV_DIR.exists(): else: python_exe = "python3" -subprocess.check_call([python_exe, "generate_samples.py"] + sys.argv[1:]) +subprocess.check_call([python_exe, "-m", "piper_sample_generator"] + sys.argv[1:]) From 275077a1a39891b008de822e5da3dafe02ecc92c Mon Sep 17 00:00:00 2001 From: Michael Hansen Date: Thu, 12 Mar 2026 15:19:01 -0500 Subject: [PATCH 15/16] Include impulses --- pyproject.toml | 3 +++ 1 file changed, 3 insertions(+) diff --git a/pyproject.toml b/pyproject.toml index 99a0bd1..76cf7d2 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -40,3 +40,6 @@ zip-safe = true [tool.setuptools.packages.find] include = ["piper_sample_generator*"] + +[tool.setuptools.package-data] +piper_sample_generator = ["impulses/*.wav"] From 2971426a55072f7d22fec416ca7800df8bd23207 Mon Sep 17 00:00:00 2001 From: Michael Hansen Date: Thu, 12 Mar 2026 15:20:12 -0500 Subject: [PATCH 16/16] Update install instructions --- README.md | 10 +--------- 1 file changed, 1 insertion(+), 9 deletions(-) diff --git a/README.md b/README.md index 4b4fd07..944d0cf 100644 --- a/README.md +++ b/README.md @@ -6,16 +6,8 @@ Supports normal [Piper voices][piper voices] or a special [generator][] that can ## Install -Create a virtual environment and install the requirements: - ``` sh -git clone https://github.com/rhasspy/piper-sample-generator.git -cd piper-sample-generator/ - -python3 -m venv .venv -source .venv/bin/activate -python3 -m pip install --upgrade pip -python3 -m pip install -e . +pip install piper-sample-generator ``` ## Piper Voices