Supertonic 2
This commit is contained in:
@@ -12,7 +12,7 @@ struct ContentView: View {
|
||||
Spacer()
|
||||
|
||||
VStack(spacing: 12) {
|
||||
Text("SupertonicTTS iOS Demo")
|
||||
Text("Supertonic 2 iOS Demo")
|
||||
.font(.title2.weight(.semibold))
|
||||
.foregroundColor(.primary)
|
||||
|
||||
@@ -44,6 +44,19 @@ struct ContentView: View {
|
||||
}
|
||||
.pickerStyle(SegmentedPickerStyle())
|
||||
.padding(.horizontal)
|
||||
|
||||
HStack(spacing: 12) {
|
||||
Text("Language")
|
||||
.font(.subheadline)
|
||||
.foregroundColor(.secondary)
|
||||
Picker("Language", selection: $vm.language) {
|
||||
ForEach(TTSService.Language.allCases, id: \.self) { lang in
|
||||
Text(lang.displayName).tag(lang)
|
||||
}
|
||||
}
|
||||
.pickerStyle(MenuPickerStyle())
|
||||
}
|
||||
.padding(.horizontal)
|
||||
}
|
||||
|
||||
HStack(spacing: 16) {
|
||||
|
||||
@@ -3,6 +3,23 @@ import OnnxRuntimeBindings
|
||||
|
||||
final class TTSService {
|
||||
enum Voice { case male, female }
|
||||
enum Language: String, CaseIterable {
|
||||
case en = "en"
|
||||
case ko = "ko"
|
||||
case es = "es"
|
||||
case pt = "pt"
|
||||
case fr = "fr"
|
||||
|
||||
var displayName: String {
|
||||
switch self {
|
||||
case .en: return "English"
|
||||
case .ko: return "한국어"
|
||||
case .es: return "Español"
|
||||
case .pt: return "Português"
|
||||
case .fr: return "Français"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private let env: ORTEnv
|
||||
private let textToSpeech: TextToSpeech
|
||||
@@ -16,13 +33,13 @@ final class TTSService {
|
||||
sampleRate = textToSpeech.sampleRate
|
||||
}
|
||||
|
||||
func synthesize(text: String, nfe: Int, voice: Voice) async throws -> URL {
|
||||
func synthesize(text: String, nfe: Int, voice: Voice, language: Language) async throws -> URL {
|
||||
// Load style for the selected voice
|
||||
let styleURL = try Self.locateVoiceStyleURL(voice: voice)
|
||||
let style = try loadVoiceStyle([styleURL.path], verbose: false)
|
||||
|
||||
// 2) Synthesize via packed TextToSpeech component
|
||||
let (wav, duration) = try textToSpeech.call(text, style, nfe)
|
||||
let (wav, duration) = try textToSpeech.call(text, language.rawValue, style, nfe)
|
||||
let audioSeconds = Double(duration)
|
||||
let wavLenSample = min(Int(Double(sampleRate) * audioSeconds), wav.count)
|
||||
let wavOut = Array(wav[0..<wavLenSample])
|
||||
|
||||
@@ -6,6 +6,7 @@ final class TTSViewModel: ObservableObject {
|
||||
@Published var text: String = "This morning, I took a walk in the park, and the sound of the birds and the breeze was so pleasant that I stopped for a long time just to listen."
|
||||
@Published var nfe: Double = 5
|
||||
@Published var voice: TTSService.Voice = .male
|
||||
@Published var language: TTSService.Language = .en
|
||||
@Published var isGenerating: Bool = false
|
||||
@Published var isPlaying: Bool = false
|
||||
@Published var errorMessage: String?
|
||||
@@ -39,7 +40,7 @@ final class TTSViewModel: ObservableObject {
|
||||
Task {
|
||||
let tic = Date()
|
||||
do {
|
||||
let url = try await service.synthesize(text: text, nfe: Int(nfe), voice: voice)
|
||||
let url = try await service.synthesize(text: text, nfe: Int(nfe), voice: voice, language: language)
|
||||
let elapsed = Date().timeIntervalSince(tic)
|
||||
let audio = audioDuration(at: url)
|
||||
await MainActor.run {
|
||||
|
||||
Reference in New Issue
Block a user