Symmetrisch zur Web-speech.ts. Neue Modalität nach Vision; TTS liefert Binär-Audio, also ein eigener Adapter (kein SSE-Stream). - ByokCapability.speak + TTS-Metadaten auf ByokProviderID (speechModels/voices/defaults); speak nur für OpenAI + Gemini - ByokSpeech.swift: byokSynthesizeSpeech-Dispatcher. OpenAI /v1/audio/speech → fertiges Format; Gemini :generateContent responseModalities AUDIO → rohes 16-bit-PCM (audio/L16) → byokPCMToWav verpackt zu WAV, byokParsePCMRate liest Rate aus MIME - ManaLLM.synthesizeSpeech(text:voice:model:provider:format:) — Facade, greift nur bei BYOK + speak-fähigem Provider (sonst nil → Server/ mana-tts); provider als Wunsch an den Vault-Resolver - 8 Tests (WAV-Header + Samples unangetastet, parsePCMRate, speak- Registry, Facade-Gating inkl. preferred), netz-/keychain-frei. swift build + swift test 47/47 grün Offen: audioguide-native StopEditor-Konsument (Xcode, Tills Hand). Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
174 lines
6.6 KiB
Swift
174 lines
6.6 KiB
Swift
import Foundation
|
|
|
|
// BYOK Text-to-Speech (Phase B: speak).
|
|
//
|
|
// Symmetrisch zur Web-Seite (`@mana/byok-providers` `speech.ts`). Anders
|
|
// als Chat/Vision (SSE-Text) liefert TTS **Binär-Audio**; der Call geht
|
|
// direkt vom Gerät zum Anbieter mit dem Key des Nutzers — der Key berührt
|
|
// mana nie.
|
|
//
|
|
// Zwei Wire-Schemata:
|
|
// - **OpenAI** (`/v1/audio/speech`) liefert ein fertiges Container-Format
|
|
// (Default mp3) → 1:1 durchgereicht.
|
|
// - **Gemini** (`:generateContent`, `responseModalities:["AUDIO"]`)
|
|
// liefert **rohes 16-bit-PCM** (`audio/L16;rate=24000`) base64-kodiert.
|
|
// Ohne Container nicht abspielbar → wir verpacken zu WAV (`byokPCMToWav`).
|
|
|
|
/// Ergebnis eines TTS-Calls: Roh-Audio-Bytes + MIME-Type.
|
|
public struct ByokSpeechResult: Sendable, Equatable {
|
|
public let audio: Data
|
|
public let mimeType: String
|
|
|
|
public init(audio: Data, mimeType: String) {
|
|
self.audio = audio
|
|
self.mimeType = mimeType
|
|
}
|
|
}
|
|
|
|
private let formatMIME: [String: String] = [
|
|
"mp3": "audio/mpeg",
|
|
"opus": "audio/opus",
|
|
"aac": "audio/aac",
|
|
"flac": "audio/flac",
|
|
"wav": "audio/wav",
|
|
]
|
|
|
|
/// OpenAI `/v1/audio/speech` → Binär-Audio (Default mp3).
|
|
func byokSynthesizeOpenAISpeech(
|
|
apiKey: String,
|
|
model: String,
|
|
text: String,
|
|
voice: String,
|
|
format: String
|
|
) async throws -> ByokSpeechResult {
|
|
var request = URLRequest(url: URL(string: "https://api.openai.com/v1/audio/speech")!)
|
|
request.httpMethod = "POST"
|
|
request.setValue("application/json", forHTTPHeaderField: "Content-Type")
|
|
request.setValue("Bearer \(apiKey)", forHTTPHeaderField: "Authorization")
|
|
request.httpBody = try JSONSerialization.data(withJSONObject: [
|
|
"model": model,
|
|
"input": text,
|
|
"voice": voice,
|
|
"response_format": format,
|
|
])
|
|
|
|
let (data, response) = try await URLSession.shared.data(for: request)
|
|
if let http = response as? HTTPURLResponse, http.statusCode >= 400 {
|
|
throw ByokError.httpError(
|
|
status: http.statusCode,
|
|
body: String(data: data.prefix(300), encoding: .utf8) ?? ""
|
|
)
|
|
}
|
|
return ByokSpeechResult(audio: data, mimeType: formatMIME[format] ?? "audio/mpeg")
|
|
}
|
|
|
|
/// Gemini-TTS via `:generateContent` mit `responseModalities:["AUDIO"]`.
|
|
///
|
|
/// Gemini liefert **rohes PCM** (signed 16-bit, mono) als base64 im
|
|
/// `inlineData`-Part, mit einem MIME wie `audio/L16;codec=pcm;rate=24000`.
|
|
/// Wir lesen die Sample-Rate aus dem MIME und wickeln die Samples in einen
|
|
/// WAV-Container. `format` wird ignoriert (Gemini gibt nur PCM); das
|
|
/// Ergebnis ist immer `audio/wav`.
|
|
func byokSynthesizeGeminiSpeech(
|
|
apiKey: String,
|
|
model: String,
|
|
text: String,
|
|
voice: String
|
|
) async throws -> ByokSpeechResult {
|
|
let urlString = "https://generativelanguage.googleapis.com/v1beta/models/\(model):generateContent?key=\(apiKey)"
|
|
var request = URLRequest(url: URL(string: urlString)!)
|
|
request.httpMethod = "POST"
|
|
request.setValue("application/json", forHTTPHeaderField: "Content-Type")
|
|
request.httpBody = try JSONSerialization.data(withJSONObject: [
|
|
"contents": [["parts": [["text": text]]]],
|
|
"generationConfig": [
|
|
"responseModalities": ["AUDIO"],
|
|
"speechConfig": ["voiceConfig": ["prebuiltVoiceConfig": ["voiceName": voice]]],
|
|
],
|
|
])
|
|
|
|
let (data, response) = try await URLSession.shared.data(for: request)
|
|
if let http = response as? HTTPURLResponse, http.statusCode >= 400 {
|
|
throw ByokError.httpError(
|
|
status: http.statusCode,
|
|
body: String(data: data.prefix(300), encoding: .utf8) ?? ""
|
|
)
|
|
}
|
|
guard let json = try? JSONSerialization.jsonObject(with: data) as? [String: Any],
|
|
let candidates = json["candidates"] as? [[String: Any]],
|
|
let content = candidates.first?["content"] as? [String: Any],
|
|
let parts = content["parts"] as? [[String: Any]]
|
|
else {
|
|
throw ByokError.malformedResponse("Gemini TTS: unerwartetes Antwort-Schema")
|
|
}
|
|
guard let inline = parts.compactMap({ $0["inlineData"] as? [String: Any] }).first,
|
|
let b64 = inline["data"] as? String,
|
|
let pcm = Data(base64Encoded: b64)
|
|
else {
|
|
throw ByokError.emptyResponse
|
|
}
|
|
let rate = byokParsePCMRate(inline["mimeType"] as? String)
|
|
return ByokSpeechResult(audio: byokPCMToWav(pcm, sampleRate: rate), mimeType: "audio/wav")
|
|
}
|
|
|
|
/// Dispatcht TTS nach Provider. Wirft, wenn der Provider kein TTS kann.
|
|
func byokSynthesizeSpeech(
|
|
_ id: ByokProviderID,
|
|
apiKey: String,
|
|
model: String,
|
|
text: String,
|
|
voice: String,
|
|
format: String
|
|
) async throws -> ByokSpeechResult {
|
|
switch id {
|
|
case .openai:
|
|
return try await byokSynthesizeOpenAISpeech(apiKey: apiKey, model: model, text: text, voice: voice, format: format)
|
|
case .gemini:
|
|
return try await byokSynthesizeGeminiSpeech(apiKey: apiKey, model: model, text: text, voice: voice)
|
|
case .anthropic, .mistral:
|
|
throw ByokError.malformedResponse("Provider \(id.displayName) unterstützt (noch) kein TTS.")
|
|
}
|
|
}
|
|
|
|
// MARK: - PCM → WAV (der testbare Kern)
|
|
|
|
/// Liest die Sample-Rate aus einem PCM-MIME wie
|
|
/// `audio/L16;codec=pcm;rate=24000`. Fällt auf 24000 (Gemini-Default)
|
|
/// zurück, wenn kein `rate=`-Parameter da ist.
|
|
func byokParsePCMRate(_ mimeType: String?) -> Int {
|
|
guard let mimeType,
|
|
let range = mimeType.range(of: #"rate=(\d+)"#, options: .regularExpression)
|
|
else { return 24000 }
|
|
let digits = mimeType[range].dropFirst("rate=".count)
|
|
return Int(digits).flatMap { $0 > 0 ? $0 : nil } ?? 24000
|
|
}
|
|
|
|
/// Verpackt rohes PCM (signed 16-bit) in einen minimalen WAV-Container
|
|
/// (44-Byte-RIFF-Header + Daten). Reine Byte-Arithmetik, netz-/
|
|
/// plattformfrei → der testbare Kern des Gemini-Adapters.
|
|
func byokPCMToWav(_ pcm: Data, sampleRate: Int, channels: Int = 1, bitsPerSample: Int = 16) -> Data {
|
|
let bytesPerSample = bitsPerSample / 8
|
|
let blockAlign = channels * bytesPerSample
|
|
let byteRate = sampleRate * blockAlign
|
|
|
|
var out = Data(capacity: 44 + pcm.count)
|
|
func ascii(_ s: String) { out.append(contentsOf: s.utf8) }
|
|
func u32(_ v: Int) { out.append(contentsOf: withUnsafeBytes(of: UInt32(v).littleEndian, Array.init)) }
|
|
func u16(_ v: Int) { out.append(contentsOf: withUnsafeBytes(of: UInt16(v).littleEndian, Array.init)) }
|
|
|
|
ascii("RIFF")
|
|
u32(36 + pcm.count) // RIFF-Chunk-Größe
|
|
ascii("WAVE")
|
|
ascii("fmt ")
|
|
u32(16) // fmt-Chunk-Größe (PCM)
|
|
u16(1) // Audio-Format = 1 (PCM)
|
|
u16(channels)
|
|
u32(sampleRate)
|
|
u32(byteRate)
|
|
u16(blockAlign)
|
|
u16(bitsPerSample)
|
|
ascii("data")
|
|
u32(pcm.count)
|
|
out.append(pcm)
|
|
return out
|
|
}
|