Supertonic 2
This commit is contained in:
@@ -19,6 +19,7 @@ namespace Supertonic
|
||||
{
|
||||
"This morning, I took a walk in the park, and the sound of the birds and the breeze was so pleasant that I stopped for a long time just to listen."
|
||||
};
|
||||
public List<string> Lang { get; set; } = new List<string> { "en" };
|
||||
public string SaveDir { get; set; } = "results";
|
||||
public bool Batch { get; set; } = false;
|
||||
}
|
||||
@@ -55,6 +56,9 @@ namespace Supertonic
|
||||
case "--text" when i + 1 < args.Length:
|
||||
result.Text = args[++i].Split('|').ToList();
|
||||
break;
|
||||
case "--lang" when i + 1 < args.Length:
|
||||
result.Lang = args[++i].Split(',').ToList();
|
||||
break;
|
||||
case "--save-dir" when i + 1 < args.Length:
|
||||
result.SaveDir = args[++i];
|
||||
break;
|
||||
@@ -76,6 +80,7 @@ namespace Supertonic
|
||||
string saveDir = parsedArgs.SaveDir;
|
||||
var voiceStylePaths = parsedArgs.VoiceStyle;
|
||||
var textList = parsedArgs.Text;
|
||||
var langList = parsedArgs.Lang;
|
||||
bool batch = parsedArgs.Batch;
|
||||
|
||||
if (voiceStylePaths.Count != textList.Count)
|
||||
@@ -101,11 +106,11 @@ namespace Supertonic
|
||||
{
|
||||
if (batch)
|
||||
{
|
||||
return textToSpeech.Batch(textList, style, totalStep, speed);
|
||||
return textToSpeech.Batch(textList, langList, style, totalStep, speed);
|
||||
}
|
||||
else
|
||||
{
|
||||
return textToSpeech.Call(textList[0], style, totalStep, speed);
|
||||
return textToSpeech.Call(textList[0], langList[0], style, totalStep, speed);
|
||||
}
|
||||
});
|
||||
|
||||
|
||||
@@ -10,6 +10,12 @@ using Microsoft.ML.OnnxRuntime.Tensors;
|
||||
|
||||
namespace Supertonic
|
||||
{
|
||||
// Available languages for multilingual TTS
|
||||
public static class Languages
|
||||
{
|
||||
public static readonly string[] Available = { "en", "ko", "es", "pt", "fr" };
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// Configuration classes
|
||||
// ============================================================================
|
||||
@@ -118,13 +124,11 @@ namespace Supertonic
|
||||
return result.ToString();
|
||||
}
|
||||
|
||||
private string PreprocessText(string text)
|
||||
private string PreprocessText(string text, string lang)
|
||||
{
|
||||
// TODO: Need advanced normalizer for better performance
|
||||
text = text.Normalize(NormalizationForm.FormKD);
|
||||
|
||||
// FIXME: this should be fixed for non-English languages
|
||||
|
||||
// Remove emojis (wide Unicode range)
|
||||
// C# doesn't support \u{...} syntax in regex, so we use character filtering instead
|
||||
text = RemoveEmojis(text);
|
||||
@@ -135,7 +139,6 @@ namespace Supertonic
|
||||
{"–", "-"}, // en dash
|
||||
{"‑", "-"}, // non-breaking hyphen
|
||||
{"—", "-"}, // em dash
|
||||
{"¯", " "}, // macron
|
||||
{"_", " "}, // underscore
|
||||
{"\u201C", "\""}, // left double quote
|
||||
{"\u201D", "\""}, // right double quote
|
||||
@@ -157,9 +160,6 @@ namespace Supertonic
|
||||
text = text.Replace(kvp.Key, kvp.Value);
|
||||
}
|
||||
|
||||
// Remove combining diacritics // FIXME: this should be fixed for non-English languages
|
||||
text = Regex.Replace(text, @"[\u0302\u0303\u0304\u0305\u0306\u0307\u0308\u030A\u030B\u030C\u0327\u0328\u0329\u032A\u032B\u032C\u032D\u032E\u032F]", "");
|
||||
|
||||
// Remove special symbols
|
||||
text = Regex.Replace(text, @"[♥☆♡©\\]", "");
|
||||
|
||||
@@ -208,6 +208,15 @@ namespace Supertonic
|
||||
text += ".";
|
||||
}
|
||||
|
||||
// Validate language
|
||||
if (!Languages.Available.Contains(lang))
|
||||
{
|
||||
throw new ArgumentException($"Invalid language: {lang}. Available: {string.Join(", ", Languages.Available)}");
|
||||
}
|
||||
|
||||
// Wrap text with language tags
|
||||
text = $"<{lang}>" + text + $"</{lang}>";
|
||||
|
||||
return text;
|
||||
}
|
||||
|
||||
@@ -221,9 +230,9 @@ namespace Supertonic
|
||||
return Helper.LengthToMask(textIdsLengths);
|
||||
}
|
||||
|
||||
public (long[][] textIds, float[][][] textMask) Call(List<string> textList)
|
||||
public (long[][] textIds, float[][][] textMask) Call(List<string> textList, List<string> langList)
|
||||
{
|
||||
var processedTexts = textList.Select(t => PreprocessText(t)).ToList();
|
||||
var processedTexts = textList.Select((t, i) => PreprocessText(t, langList[i])).ToList();
|
||||
var textIdsLengths = processedTexts.Select(t => (long)t.Length).ToArray();
|
||||
long maxLen = textIdsLengths.Max();
|
||||
|
||||
@@ -328,7 +337,7 @@ namespace Supertonic
|
||||
return (noisyLatent, latentMask);
|
||||
}
|
||||
|
||||
private (float[] wav, float[] duration) _Infer(List<string> textList, Style style, int totalStep, float speed = 1.05f)
|
||||
private (float[] wav, float[] duration) _Infer(List<string> textList, List<string> langList, Style style, int totalStep, float speed = 1.05f)
|
||||
{
|
||||
int bsz = textList.Count;
|
||||
if (bsz != style.TtlShape[0])
|
||||
@@ -337,7 +346,7 @@ namespace Supertonic
|
||||
}
|
||||
|
||||
// Process text
|
||||
var (textIds, textMask) = _textProcessor.Call(textList);
|
||||
var (textIds, textMask) = _textProcessor.Call(textList, langList);
|
||||
var textIdsShape = new long[] { bsz, textIds[0].Length };
|
||||
var textMaskShape = new long[] { bsz, 1, textMask[0][0].Length };
|
||||
|
||||
@@ -424,20 +433,21 @@ namespace Supertonic
|
||||
return (wavTensor.ToArray(), durOnnx);
|
||||
}
|
||||
|
||||
public (float[] wav, float[] duration) Call(string text, Style style, int totalStep, float speed = 1.05f, float silenceDuration = 0.3f)
|
||||
public (float[] wav, float[] duration) Call(string text, string lang, Style style, int totalStep, float speed = 1.05f, float silenceDuration = 0.3f)
|
||||
{
|
||||
if (style.TtlShape[0] != 1)
|
||||
{
|
||||
throw new ArgumentException("Single speaker text to speech only supports single style");
|
||||
}
|
||||
|
||||
var textList = Helper.ChunkText(text);
|
||||
int maxLen = lang == "ko" ? 120 : 300;
|
||||
var textList = Helper.ChunkText(text, maxLen);
|
||||
var wavCat = new List<float>();
|
||||
float durCat = 0.0f;
|
||||
|
||||
foreach (var chunk in textList)
|
||||
{
|
||||
var (wav, duration) = _Infer(new List<string> { chunk }, style, totalStep, speed);
|
||||
var (wav, duration) = _Infer(new List<string> { chunk }, new List<string> { lang }, style, totalStep, speed);
|
||||
|
||||
if (wavCat.Count == 0)
|
||||
{
|
||||
@@ -457,9 +467,9 @@ namespace Supertonic
|
||||
return (wavCat.ToArray(), new float[] { durCat });
|
||||
}
|
||||
|
||||
public (float[] wav, float[] duration) Batch(List<string> textList, Style style, int totalStep, float speed = 1.05f)
|
||||
public (float[] wav, float[] duration) Batch(List<string> textList, List<string> langList, Style style, int totalStep, float speed = 1.05f)
|
||||
{
|
||||
return _Infer(textList, style, totalStep, speed);
|
||||
return _Infer(textList, langList, style, totalStep, speed);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -4,6 +4,8 @@ This guide provides examples for running TTS inference using `ExampleONNX.cs`.
|
||||
|
||||
## 📰 Update News
|
||||
|
||||
**2026.01.06** - 🎉 **Supertonic 2** released with multilingual support! Now supports English (`en`), Korean (`ko`), Spanish (`es`), Portuguese (`pt`), and French (`fr`). [Demo](https://huggingface.co/spaces/Supertone/supertonic-2) | [Models](https://huggingface.co/Supertone/supertonic-2)
|
||||
|
||||
**2025.12.10** - Added [6 new voice styles](https://huggingface.co/Supertone/supertonic/tree/b10dbaf18b316159be75b34d24f740008fddd381) (M3, M4, M5, F3, F4, F5). See [Voices](https://supertone-inc.github.io/supertonic-py/voices/) for details
|
||||
|
||||
**2025.12.08** - Optimized ONNX models via [OnnxSlim](https://github.com/inisis/OnnxSlim) now available on [Hugging Face Models](https://huggingface.co/Supertone/supertonic)
|
||||
@@ -45,15 +47,16 @@ Process multiple voice styles and texts at once:
|
||||
```bash
|
||||
dotnet run -- \
|
||||
--voice-style assets/voice_styles/M1.json,assets/voice_styles/F1.json \
|
||||
--text "The sun sets behind the mountains, painting the sky in shades of pink and orange.|The weather is beautiful and sunny outside. A gentle breeze makes the air feel fresh and pleasant." \
|
||||
--text "The sun sets behind the mountains, painting the sky in shades of pink and orange.|오늘 아침에 공원을 산책했는데, 새소리와 바람 소리가 너무 좋아서 한참을 멈춰 서서 들었어요." \
|
||||
--lang en,ko \
|
||||
--batch
|
||||
```
|
||||
|
||||
This will:
|
||||
- Use `--batch` flag to enable batch processing mode
|
||||
- Generate speech for 2 different voice-text pairs
|
||||
- Use male voice style (M1.json) for the first text
|
||||
- Use female voice style (F1.json) for the second text
|
||||
- Use male voice style (M1.json) for the first English text
|
||||
- Use female voice style (F1.json) for the second Korean text
|
||||
- Process both samples in a single batch (automatic text chunking disabled)
|
||||
|
||||
### Example 3: High Quality Inference
|
||||
@@ -92,15 +95,18 @@ This will:
|
||||
| `--use-gpu` | flag | False | Use GPU for inference (not supported yet) |
|
||||
| `--onnx-dir` | str | `assets/onnx` | Path to ONNX model directory |
|
||||
| `--total-step` | int | 5 | Number of denoising steps (higher = better quality, slower) |
|
||||
| `--speed` | float | 1.05 | Speech speed factor (higher = faster, lower = slower) |
|
||||
| `--n-test` | int | 4 | Number of times to generate each sample |
|
||||
| `--voice-style` | str+ | `assets/voice_styles/M1.json` | Voice style file path(s) (comma-separated) |
|
||||
| `--text` | str+ | (long default text) | Text(s) to synthesize (pipe-separated: `|`) |
|
||||
| `--lang` | str+ | `en` | Language(s) for text(s): `en`, `ko`, `es`, `pt`, `fr` (comma-separated) |
|
||||
| `--save-dir` | str | `results` | Output directory |
|
||||
| `--batch` | flag | False | Enable batch mode (disables automatic text chunking) |
|
||||
|
||||
## Notes
|
||||
|
||||
- **Batch Processing**: The number of `--voice-style` files must match the number of `--text` entries
|
||||
- **Multilingual Support**: Use `--lang` to specify language(s). Available: `en` (English), `ko` (Korean), `es` (Spanish), `pt` (Portuguese), `fr` (French)
|
||||
- **Long-Form Inference**: Without `--batch` flag, long texts are automatically chunked and combined into a single audio file with natural pauses
|
||||
- **Quality vs Speed**: Higher `--total-step` values produce better quality but take longer
|
||||
- **GPU Support**: GPU mode is not supported yet
|
||||
|
||||
Reference in New Issue
Block a user