Apps ask the hub for a recognizer or a voice and get one; where it runs is the hub's decision. AiHub::start_stt / start_tts return poll-driven sessions shaped like the chat session. The Auto ladder is Whisper/Kokoro in this process (weights present, machine election), on the machine node over loopback, on a LAN node, else the OS engine; SpeechReach::Local is the "don't reach out" knob. Audio always comes back as PCM: the app owns the device. Three layers: - makepad-ai-speech is the whole speech model family, engines only. libs/voice (Whisper + Silero VAD) folds in as the `whisper` and `vad` modules next to kokoro and indextts, each a cargo feature; the Apple bridges and the Speaker/VoiceTranscriber selection leave it. - makepad-system-speech (new) is the OS speech services as blocking fns: Apple SpeechAnalyzer/AVSpeechSynthesizer via Swift, Windows.Media.Speech* on the vendored bindings, Android SpeechRecognizer/TextToSpeech through MakepadSpeech.java (API 26 floor), espeak-ng on Linux. It models the two STT shapes honestly: PCM in (Whisper, Apple) versus an engine that owns the microphone (Android, Windows), with capabilities the caller reads. - the hub grows speech sessions, in-process Whisper/Kokoro workers with the residency election, a `whisper` wire backend (stt domain, registry entry pinned to ggerganov/whisper.cpp) so a Mac can serve a Quest, and a `language` field on the generate request. Consumers: the Window voice input runs on an STT session and switches to engine-mic mode when the recognizer owns the microphone; converse's SpeechOutput is a lazily started TTS session plus a pump thread; route drops its private speech copy for converse; vj's lyrics fallback and the alignment bakes call the engines directly. Verified here: speech-roundtrip through the real sessions (Apple voice in, in-process Whisper on Metal out, 4.3% WER); system-speech-test TTS->STT verbatim; hub/converse/system-speech unit tests; msvc, aarch64-android and linux-gnu cross-checks; Java against android-34. Windows, Android and Linux bridges are compile-checked only. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
148 lines
5.4 KiB
Swift
148 lines
5.4 KiB
Swift
import AVFoundation
|
|
import Foundation
|
|
|
|
// The TTS half of makepad-system-speech's Apple bridge: AVSpeechSynthesizer
|
|
// rendered to a PCM buffer (never to a device — Makepad's audio output owns
|
|
// playback). Symbols are prefixed `mss_`.
|
|
|
|
private final class MssRendered {
|
|
var samples: [Float] = []
|
|
var sampleRate: Double = 0
|
|
}
|
|
|
|
struct MssVoice {
|
|
var id: UnsafeMutablePointer<CChar>?
|
|
var name: UnsafeMutablePointer<CChar>?
|
|
var language: UnsafeMutablePointer<CChar>?
|
|
/// 0 unknown, 1 female, 2 male (AVSpeechSynthesisVoiceGender raw values).
|
|
var gender: Int32
|
|
}
|
|
|
|
/// Render `text` to mono float PCM. `voice` is an `AVSpeechSynthesisVoice`
|
|
/// identifier or null (then `language`, a BCP-47 tag, picks the default
|
|
/// voice). `rate`/`pitch` are multipliers around 1.0. Returns null on failure;
|
|
/// release with `mss_tts_free`.
|
|
@_cdecl("mss_tts_synthesize")
|
|
public func mss_tts_synthesize(
|
|
_ text: UnsafePointer<CChar>,
|
|
_ voice: UnsafePointer<CChar>?,
|
|
_ language: UnsafePointer<CChar>,
|
|
_ rate: Float,
|
|
_ pitch: Float,
|
|
_ outLen: UnsafeMutablePointer<Int32>,
|
|
_ outRate: UnsafeMutablePointer<Float>
|
|
) -> UnsafeMutablePointer<Float>? {
|
|
outLen.pointee = 0
|
|
outRate.pointee = 0
|
|
|
|
let string = String(cString: text)
|
|
if string.trimmingCharacters(in: .whitespacesAndNewlines).isEmpty {
|
|
return nil
|
|
}
|
|
|
|
let utterance = AVSpeechUtterance(string: string)
|
|
if let voice, let selected = AVSpeechSynthesisVoice(identifier: String(cString: voice)) {
|
|
utterance.voice = selected
|
|
} else {
|
|
utterance.voice = AVSpeechSynthesisVoice(language: String(cString: language))
|
|
?? AVSpeechSynthesisVoice(language: "en-US")
|
|
}
|
|
// AVSpeechUtterance.rate is 0...1 with the default at 0.5; our 1.0 is that default.
|
|
if rate > 0 && rate.isFinite {
|
|
utterance.rate = min(max(AVSpeechUtteranceDefaultSpeechRate * rate, AVSpeechUtteranceMinimumSpeechRate),
|
|
AVSpeechUtteranceMaximumSpeechRate)
|
|
}
|
|
if pitch > 0 && pitch.isFinite {
|
|
utterance.pitchMultiplier = min(max(pitch, 0.5), 2.0)
|
|
}
|
|
|
|
let synthesizer = AVSpeechSynthesizer()
|
|
let rendered = MssRendered()
|
|
let finished = DispatchSemaphore(value: 0)
|
|
var signalled = false
|
|
|
|
// Buffers arrive on an internal queue; a zero-length buffer terminates the run.
|
|
synthesizer.write(utterance) { buffer in
|
|
guard let pcm = buffer as? AVAudioPCMBuffer else { return }
|
|
let frames = Int(pcm.frameLength)
|
|
if frames == 0 {
|
|
if !signalled {
|
|
signalled = true
|
|
finished.signal()
|
|
}
|
|
return
|
|
}
|
|
rendered.sampleRate = pcm.format.sampleRate
|
|
if let channels = pcm.floatChannelData {
|
|
rendered.samples.append(contentsOf: UnsafeBufferPointer(start: channels[0], count: frames))
|
|
} else if let channels = pcm.int16ChannelData {
|
|
let source = UnsafeBufferPointer(start: channels[0], count: frames)
|
|
rendered.samples.append(contentsOf: source.map { Float($0) / 32768.0 })
|
|
}
|
|
}
|
|
|
|
// `write` delivers its buffers through the MAIN run loop, whichever thread
|
|
// called it (pumping the caller's own loop was tried and delivers
|
|
// nothing). Every UI app pumps main, so a worker just waits on the
|
|
// semaphore; a caller ON main must pump instead of blocking, or it
|
|
// deadlocks itself. A headless tool calling from a worker must keep its
|
|
// main thread in CFRunLoopRun — see makepad-ai-hub's speech-roundtrip.
|
|
if Thread.isMainThread {
|
|
let deadline = Date().addingTimeInterval(30)
|
|
while !signalled, Date() < deadline {
|
|
RunLoop.current.run(mode: .default, before: Date().addingTimeInterval(0.02))
|
|
}
|
|
} else {
|
|
_ = finished.wait(timeout: .now() + 30)
|
|
}
|
|
withExtendedLifetime(synthesizer) {}
|
|
|
|
if rendered.samples.isEmpty || rendered.sampleRate <= 0 {
|
|
return nil
|
|
}
|
|
|
|
let count = rendered.samples.count
|
|
let out = UnsafeMutablePointer<Float>.allocate(capacity: count)
|
|
rendered.samples.withUnsafeBufferPointer { source in
|
|
out.initialize(from: source.baseAddress!, count: count)
|
|
}
|
|
outLen.pointee = Int32(count)
|
|
outRate.pointee = Float(rendered.sampleRate)
|
|
return out
|
|
}
|
|
|
|
@_cdecl("mss_tts_free")
|
|
public func mss_tts_free(_ ptr: UnsafeMutablePointer<Float>?) {
|
|
ptr?.deallocate()
|
|
}
|
|
|
|
/// Installed voices as an owned array of `MssVoice`; release with
|
|
/// `mss_tts_free_voices`.
|
|
@_cdecl("mss_tts_voices")
|
|
public func mss_tts_voices(_ outCount: UnsafeMutablePointer<Int32>) -> OpaquePointer? {
|
|
let voices = AVSpeechSynthesisVoice.speechVoices()
|
|
outCount.pointee = Int32(voices.count)
|
|
if voices.isEmpty { return nil }
|
|
let ptr = UnsafeMutablePointer<MssVoice>.allocate(capacity: voices.count)
|
|
for (i, v) in voices.enumerated() {
|
|
ptr[i] = MssVoice(
|
|
id: strdup(v.identifier),
|
|
name: strdup(v.name),
|
|
language: strdup(v.language),
|
|
gender: Int32(v.gender.rawValue)
|
|
)
|
|
}
|
|
return OpaquePointer(ptr)
|
|
}
|
|
|
|
@_cdecl("mss_tts_free_voices")
|
|
public func mss_tts_free_voices(_ ptr: OpaquePointer?, _ count: Int32) {
|
|
guard let rawPtr = ptr else { return }
|
|
let typed = UnsafeMutablePointer<MssVoice>(rawPtr)
|
|
for i in 0..<Int(count) {
|
|
if let s = typed[i].id { free(s) }
|
|
if let s = typed[i].name { free(s) }
|
|
if let s = typed[i].language { free(s) }
|
|
}
|
|
typed.deallocate()
|
|
}
|