mana-swift-llm/Sources/ManaLLM/Byok/ByokSpeech.swift
till 069e663847 BYOK TTS (Phase B): native synthesizeSpeech — OpenAI + Gemini
Symmetrisch zur Web-speech.ts. Neue Modalität nach Vision; TTS liefert
Binär-Audio, also ein eigener Adapter (kein SSE-Stream).

- ByokCapability.speak + TTS-Metadaten auf ByokProviderID
  (speechModels/voices/defaults); speak nur für OpenAI + Gemini
- ByokSpeech.swift: byokSynthesizeSpeech-Dispatcher. OpenAI
  /v1/audio/speech → fertiges Format; Gemini :generateContent
  responseModalities AUDIO → rohes 16-bit-PCM (audio/L16) →
  byokPCMToWav verpackt zu WAV, byokParsePCMRate liest Rate aus MIME
- ManaLLM.synthesizeSpeech(text:voice:model:provider:format:) — Facade,
  greift nur bei BYOK + speak-fähigem Provider (sonst nil → Server/
  mana-tts); provider als Wunsch an den Vault-Resolver
- 8 Tests (WAV-Header + Samples unangetastet, parsePCMRate, speak-
  Registry, Facade-Gating inkl. preferred), netz-/keychain-frei.
  swift build + swift test 47/47 grün

Offen: audioguide-native StopEditor-Konsument (Xcode, Tills Hand).

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-06-05 17:43:43 +02:00

174 lines
6.6 KiB
Swift

import Foundation
// BYOK Text-to-Speech (Phase B: speak).
//
// Symmetrisch zur Web-Seite (`@mana/byok-providers` `speech.ts`). Anders
// als Chat/Vision (SSE-Text) liefert TTS **Binär-Audio**; der Call geht
// direkt vom Gerät zum Anbieter mit dem Key des Nutzers — der Key berührt
// mana nie.
//
// Zwei Wire-Schemata:
// - **OpenAI** (`/v1/audio/speech`) liefert ein fertiges Container-Format
// (Default mp3) → 1:1 durchgereicht.
// - **Gemini** (`:generateContent`, `responseModalities:["AUDIO"]`)
// liefert **rohes 16-bit-PCM** (`audio/L16;rate=24000`) base64-kodiert.
// Ohne Container nicht abspielbar → wir verpacken zu WAV (`byokPCMToWav`).
/// Ergebnis eines TTS-Calls: Roh-Audio-Bytes + MIME-Type.
public struct ByokSpeechResult: Sendable, Equatable {
public let audio: Data
public let mimeType: String
public init(audio: Data, mimeType: String) {
self.audio = audio
self.mimeType = mimeType
}
}
private let formatMIME: [String: String] = [
"mp3": "audio/mpeg",
"opus": "audio/opus",
"aac": "audio/aac",
"flac": "audio/flac",
"wav": "audio/wav",
]
/// OpenAI `/v1/audio/speech` → Binär-Audio (Default mp3).
func byokSynthesizeOpenAISpeech(
apiKey: String,
model: String,
text: String,
voice: String,
format: String
) async throws -> ByokSpeechResult {
var request = URLRequest(url: URL(string: "https://api.openai.com/v1/audio/speech")!)
request.httpMethod = "POST"
request.setValue("application/json", forHTTPHeaderField: "Content-Type")
request.setValue("Bearer \(apiKey)", forHTTPHeaderField: "Authorization")
request.httpBody = try JSONSerialization.data(withJSONObject: [
"model": model,
"input": text,
"voice": voice,
"response_format": format,
])
let (data, response) = try await URLSession.shared.data(for: request)
if let http = response as? HTTPURLResponse, http.statusCode >= 400 {
throw ByokError.httpError(
status: http.statusCode,
body: String(data: data.prefix(300), encoding: .utf8) ?? ""
)
}
return ByokSpeechResult(audio: data, mimeType: formatMIME[format] ?? "audio/mpeg")
}
/// Gemini-TTS via `:generateContent` mit `responseModalities:["AUDIO"]`.
///
/// Gemini liefert **rohes PCM** (signed 16-bit, mono) als base64 im
/// `inlineData`-Part, mit einem MIME wie `audio/L16;codec=pcm;rate=24000`.
/// Wir lesen die Sample-Rate aus dem MIME und wickeln die Samples in einen
/// WAV-Container. `format` wird ignoriert (Gemini gibt nur PCM); das
/// Ergebnis ist immer `audio/wav`.
func byokSynthesizeGeminiSpeech(
apiKey: String,
model: String,
text: String,
voice: String
) async throws -> ByokSpeechResult {
let urlString = "https://generativelanguage.googleapis.com/v1beta/models/\(model):generateContent?key=\(apiKey)"
var request = URLRequest(url: URL(string: urlString)!)
request.httpMethod = "POST"
request.setValue("application/json", forHTTPHeaderField: "Content-Type")
request.httpBody = try JSONSerialization.data(withJSONObject: [
"contents": [["parts": [["text": text]]]],
"generationConfig": [
"responseModalities": ["AUDIO"],
"speechConfig": ["voiceConfig": ["prebuiltVoiceConfig": ["voiceName": voice]]],
],
])
let (data, response) = try await URLSession.shared.data(for: request)
if let http = response as? HTTPURLResponse, http.statusCode >= 400 {
throw ByokError.httpError(
status: http.statusCode,
body: String(data: data.prefix(300), encoding: .utf8) ?? ""
)
}
guard let json = try? JSONSerialization.jsonObject(with: data) as? [String: Any],
let candidates = json["candidates"] as? [[String: Any]],
let content = candidates.first?["content"] as? [String: Any],
let parts = content["parts"] as? [[String: Any]]
else {
throw ByokError.malformedResponse("Gemini TTS: unerwartetes Antwort-Schema")
}
guard let inline = parts.compactMap({ $0["inlineData"] as? [String: Any] }).first,
let b64 = inline["data"] as? String,
let pcm = Data(base64Encoded: b64)
else {
throw ByokError.emptyResponse
}
let rate = byokParsePCMRate(inline["mimeType"] as? String)
return ByokSpeechResult(audio: byokPCMToWav(pcm, sampleRate: rate), mimeType: "audio/wav")
}
/// Dispatcht TTS nach Provider. Wirft, wenn der Provider kein TTS kann.
func byokSynthesizeSpeech(
_ id: ByokProviderID,
apiKey: String,
model: String,
text: String,
voice: String,
format: String
) async throws -> ByokSpeechResult {
switch id {
case .openai:
return try await byokSynthesizeOpenAISpeech(apiKey: apiKey, model: model, text: text, voice: voice, format: format)
case .gemini:
return try await byokSynthesizeGeminiSpeech(apiKey: apiKey, model: model, text: text, voice: voice)
case .anthropic, .mistral:
throw ByokError.malformedResponse("Provider \(id.displayName) unterstützt (noch) kein TTS.")
}
}
// MARK: - PCM → WAV (der testbare Kern)
/// Liest die Sample-Rate aus einem PCM-MIME wie
/// `audio/L16;codec=pcm;rate=24000`. Fällt auf 24000 (Gemini-Default)
/// zurück, wenn kein `rate=`-Parameter da ist.
func byokParsePCMRate(_ mimeType: String?) -> Int {
guard let mimeType,
let range = mimeType.range(of: #"rate=(\d+)"#, options: .regularExpression)
else { return 24000 }
let digits = mimeType[range].dropFirst("rate=".count)
return Int(digits).flatMap { $0 > 0 ? $0 : nil } ?? 24000
}
/// Verpackt rohes PCM (signed 16-bit) in einen minimalen WAV-Container
/// (44-Byte-RIFF-Header + Daten). Reine Byte-Arithmetik, netz-/
/// plattformfrei → der testbare Kern des Gemini-Adapters.
func byokPCMToWav(_ pcm: Data, sampleRate: Int, channels: Int = 1, bitsPerSample: Int = 16) -> Data {
let bytesPerSample = bitsPerSample / 8
let blockAlign = channels * bytesPerSample
let byteRate = sampleRate * blockAlign
var out = Data(capacity: 44 + pcm.count)
func ascii(_ s: String) { out.append(contentsOf: s.utf8) }
func u32(_ v: Int) { out.append(contentsOf: withUnsafeBytes(of: UInt32(v).littleEndian, Array.init)) }
func u16(_ v: Int) { out.append(contentsOf: withUnsafeBytes(of: UInt16(v).littleEndian, Array.init)) }
ascii("RIFF")
u32(36 + pcm.count) // RIFF-Chunk-Größe
ascii("WAVE")
ascii("fmt ")
u32(16) // fmt-Chunk-Größe (PCM)
u16(1) // Audio-Format = 1 (PCM)
u16(channels)
u32(sampleRate)
u32(byteRate)
u16(blockAlign)
u16(bitsPerSample)
ascii("data")
u32(pcm.count)
out.append(pcm)
return out
}