diff --git a/FirebaseAI/CHANGELOG.md b/FirebaseAI/CHANGELOG.md index a12b80423ee..3cbe0d34e39 100644 --- a/FirebaseAI/CHANGELOG.md +++ b/FirebaseAI/CHANGELOG.md @@ -1,4 +1,9 @@ # Unreleased +- [feature] **Public Preview**: Added `SpeechMetadata` support to `TextPart` + for Gemini text-to-speech (TTS) models, enabling turn-level speaker routing + in multi-speaker synthesis and sustained speech delivery styling. See the + [speech generation guide](https://firebase.google.com/docs/ai-logic/generate-speech) + for more details. - [fixed] Fixed crashes when a chat session's history contained parts with unrecognized data or code execution parts received from the server, when creating a `ModelContent` with an unsupported `Part` type, and when decoding a diff --git a/FirebaseAI/Sources/History.swift b/FirebaseAI/Sources/History.swift index 394262fe511..ed71ab93194 100644 --- a/FirebaseAI/Sources/History.swift +++ b/FirebaseAI/Sources/History.swift @@ -63,7 +63,9 @@ final class History: Sendable { // Loop through all the parts, aggregating the text. for part in chunks.flatMap({ $0.internalParts }) { // Only text parts may be combined. - if case let .text(text) = part.data, part.thoughtSignature == nil { + if case let .text(text) = part.data, + part.thoughtSignature == nil, + part.speechMetadata == nil { // Thought summaries must not be combined with regular text. if part.isThought ?? false { // If we were combining regular text, flush it before handling "thoughts". @@ -79,7 +81,8 @@ final class History: Sendable { combinedText += text } } else { - // This is a non-combinable part (not text), flush any pending text. + // This is a non-combinable part (non-text, signed thought, or speech metadata), flush + // any pending text. flush() parts.append(part) } diff --git a/FirebaseAI/Sources/ModelContent.swift b/FirebaseAI/Sources/ModelContent.swift index 57613fa96ce..3d458e38c1c 100644 --- a/FirebaseAI/Sources/ModelContent.swift +++ b/FirebaseAI/Sources/ModelContent.swift @@ -58,8 +58,14 @@ struct InternalPart: Equatable, Sendable { let thoughtSignature: String? - init(_ data: OneOfData, isThought: Bool?, thoughtSignature: String?) { + let speechMetadata: SpeechMetadata? + + init(_ data: OneOfData, + speechMetadata: SpeechMetadata? = nil, + isThought: Bool?, + thoughtSignature: String?) { self.data = data + self.speechMetadata = speechMetadata self.isThought = isThought self.thoughtSignature = thoughtSignature } @@ -78,7 +84,12 @@ public struct ModelContent: Equatable, Sendable { return internalParts.compactMap { part -> (any Part)? in switch part.data { case let .text(text): - return TextPart(text, isThought: part.isThought, thoughtSignature: part.thoughtSignature) + return TextPart( + text, + speechMetadata: part.speechMetadata, + isThought: part.isThought, + thoughtSignature: part.thoughtSignature + ) case let .inlineData(inlineData): return InlineDataPart( inlineData, isThought: part.isThought, thoughtSignature: part.thoughtSignature @@ -126,6 +137,7 @@ public struct ModelContent: Equatable, Sendable { case let textPart as TextPart: convertedParts.append(InternalPart( .text(textPart.text), + speechMetadata: textPart.speechMetadata, isThought: textPart._isThought, thoughtSignature: textPart.thoughtSignature )) @@ -215,6 +227,7 @@ extension InternalPart: Codable { enum CodingKeys: String, CodingKey { case isThought = "thought" case thoughtSignature + case speechMetadata } public func encode(to encoder: Encoder) throws { @@ -227,6 +240,7 @@ extension InternalPart: Codable { var container = encoder.container(keyedBy: CodingKeys.self) try container.encodeIfPresent(isThought, forKey: .isThought) try container.encodeIfPresent(thoughtSignature, forKey: .thoughtSignature) + try container.encodeIfPresent(speechMetadata, forKey: .speechMetadata) } public init(from decoder: Decoder) throws { @@ -241,6 +255,7 @@ extension InternalPart: Codable { let container = try decoder.container(keyedBy: CodingKeys.self) isThought = try container.decodeIfPresent(Bool.self, forKey: .isThought) thoughtSignature = try container.decodeIfPresent(String.self, forKey: .thoughtSignature) + speechMetadata = try container.decodeIfPresent(SpeechMetadata.self, forKey: .speechMetadata) } } diff --git a/FirebaseAI/Sources/Types/Public/MultiSpeakerVoiceConfig.swift b/FirebaseAI/Sources/Types/Public/MultiSpeakerVoiceConfig.swift index 2a65aaa3909..b008a4bcc67 100644 --- a/FirebaseAI/Sources/Types/Public/MultiSpeakerVoiceConfig.swift +++ b/FirebaseAI/Sources/Types/Public/MultiSpeakerVoiceConfig.swift @@ -14,15 +14,22 @@ import Foundation -/// Configuration for a multi-speaker audio generation setup. +/// **[Public Preview]** Configuration for a multi-speaker audio generation setup. /// -/// **Public Preview**: This API is a public preview and may be subject to change. +/// > Warning: This API is a public preview and may be subject to change. /// /// Enables the model to generate audio containing multiple distinct speakers, alternating voices -/// dynamically based on speaker labels in the prompt. +/// dynamically based on the `speaker` specified in each turn's ``SpeechMetadata``. /// -/// > Warning: Multi-speaker configurations are not currently supported by the Live API (e.g., -/// > `LiveGenerationConfig`). +/// > Important: When using multi-speaker generation, every ``TextPart`` in the request prompt must +/// > include ``SpeechMetadata`` with a `speaker` matching one of the configured speakers. Omitting +/// > `speaker` in a multi-speaker request results in a backend error. +/// +/// > Warning: Multi-speaker configurations are not currently supported by the Live API, such as +/// > ``LiveGenerationConfig``. +/// +/// For more details, see the +/// [multi-speaker guide](https://firebase.google.com/docs/ai-logic/generate-speech#multi-speaker). public struct MultiSpeakerVoiceConfig: Sendable { let multiSpeakerVoiceConfig: ProtoMultiSpeakerVoiceConfig @@ -30,7 +37,7 @@ public struct MultiSpeakerVoiceConfig: Sendable { self.multiSpeakerVoiceConfig = multiSpeakerVoiceConfig } - /// Creates a configuration for the multi-speaker setup. + /// Creates a multi-speaker voice configuration. /// /// - Parameters: /// - speakerVoiceConfigs: A list of voice configurations for the participating speakers. diff --git a/FirebaseAI/Sources/Types/Public/Part.swift b/FirebaseAI/Sources/Types/Public/Part.swift index 89af955d39d..6f4323f3b72 100644 --- a/FirebaseAI/Sources/Types/Public/Part.swift +++ b/FirebaseAI/Sources/Types/Public/Part.swift @@ -31,18 +31,49 @@ public struct TextPart: Part { /// Text value. public let text: String + /// **[Public Preview]** Optional speech metadata configuring the speaker and delivery style for + /// speech synthesis. + /// + /// When using a Gemini text-to-speech model, this metadata controls turn-level delivery style + /// and speaker assignment for multi-speaker audio generation. + /// + /// > Important: When using a multi-speaker configuration, `speechMetadata` with a matching + /// > `speaker` is required on every ``TextPart``. Omitting `speaker` in a multi-speaker request + /// > results in a backend error. + /// + /// For more details, see the + /// [Text-to-speech guide](https://firebase.google.com/docs/ai-logic/generate-speech). + public let speechMetadata: SpeechMetadata? + public var isThought: Bool { _isThought ?? false } let thoughtSignature: String? let _isThought: Bool? + /// Creates a text part with a string value. + /// + /// - Parameter text: The text string. For speech generation, this represents the verbatim + /// transcript to be synthesized. public init(_ text: String) { - self.init(text, isThought: nil, thoughtSignature: nil) + self.init(text, speechMetadata: nil, isThought: nil, thoughtSignature: nil) + } + + /// Creates a text part with a string value and optional speech metadata. + /// + /// - Parameters: + /// - text: The text string. For speech generation, this represents the verbatim transcript + /// to be synthesized. + /// - speechMetadata: Optional ``SpeechMetadata`` configuring the speaker and delivery style for + /// text-to-speech generation. Defaults to `nil`. + public init(_ text: String, speechMetadata: SpeechMetadata? = nil) { + self.init(text, speechMetadata: speechMetadata, isThought: nil, thoughtSignature: nil) } - init(_ text: String, isThought: Bool?, thoughtSignature: String?) { + init(_ text: String, speechMetadata: SpeechMetadata? = nil, isThought: Bool?, + thoughtSignature: String?) { self.text = text + self.speechMetadata = speechMetadata _isThought = isThought self.thoughtSignature = thoughtSignature } diff --git a/FirebaseAI/Sources/Types/Public/SpeakerVoiceConfig.swift b/FirebaseAI/Sources/Types/Public/SpeakerVoiceConfig.swift index 32cfbd7cdff..efc590a4e50 100644 --- a/FirebaseAI/Sources/Types/Public/SpeakerVoiceConfig.swift +++ b/FirebaseAI/Sources/Types/Public/SpeakerVoiceConfig.swift @@ -14,9 +14,13 @@ import Foundation -/// Configures a speaker with a unique name/identifier and a specific voice. +/// **[Public Preview]** Configuration pairing a speaker name or identifier with a preset voice. /// -/// **Public Preview**: This API is a public preview and may be subject to change. +/// > Warning: This API is a public preview and may be subject to change. +/// +/// This configuration pairs a speaker name (such as `"Alice"` or `"Joe"`) with a preset voice name. +/// The `speaker` name defined here must match the `speaker` string specified in ``SpeechMetadata`` +/// on each dialogue turn's ``TextPart`` in a multi-speaker request. public struct SpeakerVoiceConfig: Sendable { let speakerVoiceConfig: ProtoSpeakerVoiceConfig @@ -27,11 +31,12 @@ public struct SpeakerVoiceConfig: Sendable { /// Creates a configuration for a speaker using a voice name. /// /// - Parameters: - /// - speaker: The unique name/identifier of the speaker (e.g., `"Alice"`). + /// - speaker: The unique name or identifier of the speaker (for example, `"Alice"`). This name + /// must be passed to ``SpeechMetadata/init(speaker:style:)`` for this speaker's turns. /// - voiceName: The name of the preset voice to assign to this speaker. /// - /// Find the list of supported voices at - /// https://firebase.google.com/docs/ai-logic/generate-speech#supported-voices-and-languages + /// For a list of available voices, see the documentation on + /// [voices](https://firebase.google.com/docs/ai-logic/generate-speech#response-voices). public init(speaker: String, voiceName: String) { self.init( ProtoSpeakerVoiceConfig( diff --git a/FirebaseAI/Sources/Types/Public/SpeechConfig.swift b/FirebaseAI/Sources/Types/Public/SpeechConfig.swift index 2dc4730bbb7..b1497b172a6 100644 --- a/FirebaseAI/Sources/Types/Public/SpeechConfig.swift +++ b/FirebaseAI/Sources/Types/Public/SpeechConfig.swift @@ -14,12 +14,18 @@ import Foundation -/// Speech configuration class for controlling the model's speech and audio generation behaviors. +/// **[Public Preview]** Configuration for model speech and audio generation behaviors. /// -/// **Public Preview**: This API is a public preview and may be subject to change. +/// > Warning: This API is a public preview and may be subject to change. /// -/// This allows you to configure the voice properties (single-speaker OR multi-speaker setup) and -/// language preferences when requesting the model to generate spoken responses. +/// Configures voice properties (single-speaker or multi-speaker setup) and language preferences +/// when requesting the model to generate spoken responses. +/// +/// For turn-level speech delivery control (such as emotion, pacing, and whispering), attach +/// ``SpeechMetadata`` to individual ``TextPart`` instances in your request prompt. +/// +/// For more details on speech generation, see the +/// [Text-to-speech guide](https://firebase.google.com/docs/ai-logic/generate-speech). public struct SpeechConfig: Sendable { let speechConfig: ProtoSpeechConfig @@ -32,13 +38,13 @@ public struct SpeechConfig: Sendable { /// - Parameters: /// - voiceName: The name of the prebuilt voice to be used for the model's speech response. /// - /// To learn more about the available voices, see the docs on - /// [Voice options](https://ai.google.dev/gemini-api/docs/speech-generation#voices)\. + /// For available voices, see the documentation on + /// [voices](https://firebase.google.com/docs/ai-logic/generate-speech#response-voices). /// - languageCode: BCP-47 language code to use when parsing text sent from the client, instead /// of audio. By default, the model will attempt to detect the input language automatically. /// - /// To learn which codes are supported, see the docs on - /// [Supported languages](https://ai.google.dev/gemini-api/docs/speech-generation#languages)\. + /// For supported language codes, see the documentation on + /// [languages](https://firebase.google.com/docs/ai-logic/generate-speech#languages). public init(voiceName: String, languageCode: String? = nil) { self.init( ProtoSpeechConfig( @@ -50,13 +56,25 @@ public struct SpeechConfig: Sendable { /// Creates a new ``SpeechConfig`` value for a multi-speaker setup. /// - /// > Warning: Multi-speaker configurations are not currently supported by the Live API (e.g., - /// > `LiveGenerationConfig`). + /// > Warning: Multi-speaker configurations are not currently supported by the Live API, such as + /// > ``LiveGenerationConfig``. /// /// - Parameters: /// - multiSpeakerVoiceConfig: The configuration detailing multiple speakers and their /// corresponding voices. + /// + /// > Important: When using a multi-speaker configuration, each dialogue turn in the + /// > request prompt must be passed as a separate ``TextPart`` with ``SpeechMetadata`` + /// > specifying a `speaker` matching one of the configured speakers. `speaker` is required + /// > on every part in a multi-speaker request. + /// + /// See the documentation on + /// [multi-speaker](https://firebase.google.com/docs/ai-logic/generate-speech#multi-speaker) + /// for more details. /// - languageCode: BCP-47 language code to use when parsing text sent from the client. + /// + /// For supported language codes, see the documentation on + /// [languages](https://firebase.google.com/docs/ai-logic/generate-speech#languages). public init(multiSpeakerVoiceConfig: MultiSpeakerVoiceConfig, languageCode: String? = nil) { self.init( ProtoSpeechConfig( diff --git a/FirebaseAI/Sources/Types/Public/SpeechMetadata.swift b/FirebaseAI/Sources/Types/Public/SpeechMetadata.swift new file mode 100644 index 00000000000..2c9562866d0 --- /dev/null +++ b/FirebaseAI/Sources/Types/Public/SpeechMetadata.swift @@ -0,0 +1,71 @@ +// Copyright 2026 Google LLC +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +/// **[Public Preview]** Turn-level speech synthesis metadata for text content in a model request. +/// +/// > Warning: This API is a public preview and may be subject to change. +/// +/// When generating audio using Gemini text-to-speech (TTS) models, the input text in a +/// ``TextPart`` is treated strictly as a verbatim transcript. ``SpeechMetadata`` attaches +/// structured turn-level instructions to control speaker assignment and speech delivery. +/// +/// For more details on speech generation, see the +/// [Text-to-speech guide](https://firebase.google.com/docs/ai-logic/generate-speech). +public struct SpeechMetadata: Sendable, Equatable, Hashable { + /// The unique name or identifier of the speaker for multi-speaker synthesis. + /// + /// > Important: When using a multi-speaker configuration, `speaker` is required on every + /// > ``TextPart``. Omitting `speaker` in a multi-speaker request results in a backend error. + let speaker: String? + + /// The sustained delivery style instruction for speech synthesis. + let style: String? + + /// Creates speech metadata with an optional speaker identifier and delivery style. + /// + /// - Parameters: + /// - speaker: The unique name or identifier of the speaker for multi-speaker synthesis + /// (for example, `"Joe"`). Must match a speaker name configured in ``SpeakerVoiceConfig``. + /// + /// > Important: When using a multi-speaker configuration, `speaker` is required on every + /// > dialogue turn. Pass each speaker turn as a separate ``TextPart`` with a matching + /// > `speaker` name. + /// + /// See the documentation on + /// [multi-speaker](https://firebase.google.com/docs/ai-logic/generate-speech#multi-speaker) + /// for more details. Defaults to `nil`. + /// - style: A natural-language description of the sustained delivery style, emotion, + /// prosody, pacing, or volume across the entire turn (for example, + /// `"cheerful and friendly"` or `"whispering urgently"`). + /// + /// Instructions should be partitioned based on scope: + /// - **Sustained turn-level delivery (`style`)**: Put attributes that apply across an + /// entire dialogue turn into `style` (for example, `"whispering"`, + /// `"cheerful and friendly"`, `"speaking slowly"`, `"out of breath"`, or `"sarcastic"`). + /// - **Point-in-time events (inline vocal tags)**: Place momentary non-speech vocalizations + /// or pauses directly inside the transcript text using angle brackets (for example, + /// ``, ``, ``, ``, or ``), rather than in + /// `style`. + /// + /// See [audio tags](https://firebase.google.com/docs/ai-logic/generate-speech#audio-tags) + /// for more details. Defaults to `nil`. + public init(speaker: String? = nil, style: String? = nil) { + self.speaker = speaker + self.style = style + } +} + +// MARK: - Codable Conformance + +extension SpeechMetadata: Codable {} diff --git a/FirebaseAI/Tests/TestApp/Sources/Constants.swift b/FirebaseAI/Tests/TestApp/Sources/Constants.swift index 41805ef6333..a9c5d6a7623 100644 --- a/FirebaseAI/Tests/TestApp/Sources/Constants.swift +++ b/FirebaseAI/Tests/TestApp/Sources/Constants.swift @@ -29,6 +29,7 @@ public enum ModelNames { public static let gemini2_5_Pro = "gemini-2.5-pro" public static let gemini3_1_FlashLite = "gemini-3.1-flash-lite" public static let gemini3_1_FlashImage = "gemini-3.1-flash-image" - public static let gemini3_1_FlashTTSPreview = "gemini-3.1-flash-tts-preview" + public static let gemini3_8_FlashTTS = "gemini-3.8-flash-tts" + public static let gemini3_8_FlashLiteTTS = "gemini-3.8-flash-lite-tts" public static let gemma4_31B = "gemma-4-31b-it" } diff --git a/FirebaseAI/Tests/TestApp/Tests/Integration/GenerateContentIntegrationTests.swift b/FirebaseAI/Tests/TestApp/Tests/Integration/GenerateContentIntegrationTests.swift index 2f563480af2..4e5ff0d7739 100644 --- a/FirebaseAI/Tests/TestApp/Tests/Integration/GenerateContentIntegrationTests.swift +++ b/FirebaseAI/Tests/TestApp/Tests/Integration/GenerateContentIntegrationTests.swift @@ -740,95 +740,6 @@ struct GenerateContentIntegrationTests { return String(describing: underlyingError).contains("Firebase App Check token is invalid") } } - - @Test(arguments: InstanceConfig.defaultConfigs) - func generateContent_speechConfig(_ config: InstanceConfig) async throws { - let model = FirebaseAI.componentInstance(config).generativeModel( - modelName: ModelNames.gemini3_1_FlashTTSPreview, - generationConfig: GenerationConfig( - responseModalities: [.audio], - speechConfig: SpeechConfig(voiceName: "Charon", languageCode: "en-US") - ), - safetySettings: safetySettings - ) - let response = try await model.generateContent("Hello") - let candidate = try #require(response.candidates.first) - #expect(candidate.finishReason == .stop) - } - - @Test(arguments: InstanceConfig.defaultConfigs) - func generateContent_speechConfig_multiSpeaker(_ config: InstanceConfig) async throws { - let model = FirebaseAI.componentInstance(config).generativeModel( - modelName: ModelNames.gemini3_1_FlashTTSPreview, - generationConfig: GenerationConfig( - responseModalities: [.audio], - speechConfig: SpeechConfig( - multiSpeakerVoiceConfig: MultiSpeakerVoiceConfig( - speakerVoiceConfigs: [ - SpeakerVoiceConfig( - speaker: "Speaker1", - voiceName: "Puck" - ), - SpeakerVoiceConfig( - speaker: "Speaker2", - voiceName: "Charon" - ), - ] - ), - languageCode: "en-US" - ) - ), - safetySettings: safetySettings - ) - let response = try await model.generateContent("Hello") - let candidate = try #require(response.candidates.first) - #expect(candidate.finishReason == .stop) - } - - @Test(arguments: InstanceConfig.defaultConfigs) - func generateContent_speechConfig_multiSpeaker_invalidSize(_ config: InstanceConfig) async throws { - let model = FirebaseAI.componentInstance(config).generativeModel( - modelName: ModelNames.gemini3_1_FlashTTSPreview, - generationConfig: GenerationConfig( - responseModalities: [.audio], - speechConfig: SpeechConfig( - multiSpeakerVoiceConfig: MultiSpeakerVoiceConfig( - speakerVoiceConfigs: [ - SpeakerVoiceConfig( - speaker: "Speaker1", - voiceName: "Puck" - ), - SpeakerVoiceConfig( - speaker: "Speaker2", - voiceName: "Charon" - ), - SpeakerVoiceConfig( - speaker: "Speaker3", - voiceName: "Aoede" - ), - ] - ), - languageCode: "en-US" - ) - ), - safetySettings: safetySettings - ) - do { - _ = try await model.generateContent("Hello") - Issue.record("Expected an error from the backend for invalid multi-speaker list size.") - } catch { - guard let error = error as? GenerateContentError else { - Issue.record("Expected GenerateContentError; got \(error.self).") - throw error - } - guard case let .internalError(underlyingError) = error else { - Issue.record("Expected internalError; got \(error.self).") - throw error - } - #expect(String(describing: underlyingError) - .contains("the number of speaker_voice_configs must equal 2")) - } - } } extension TextPart { diff --git a/FirebaseAI/Tests/TestApp/Tests/Integration/TextToSpeechTests.swift b/FirebaseAI/Tests/TestApp/Tests/Integration/TextToSpeechTests.swift new file mode 100644 index 00000000000..7e2ade71776 --- /dev/null +++ b/FirebaseAI/Tests/TestApp/Tests/Integration/TextToSpeechTests.swift @@ -0,0 +1,246 @@ +// Copyright 2026 Google LLC +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +import FirebaseAILogic +import FirebaseAITestApp +import Testing + +@Suite(.serialized) +struct TextToSpeechTests { + private static let ttsModels = [ + ModelNames.gemini3_8_FlashTTS, + ModelNames.gemini3_8_FlashLiteTTS, + ] + private let unaryMIMEType = "audio/wav" + private let streamingMIMEType = "audio/l16; rate=24000; channels=1" + + // MARK: - Single-Speaker Tests + + @Test(arguments: InstanceConfig.defaultConfigs, ttsModels) + func singleSpeaker_withTurnMetadata(_ config: InstanceConfig, modelName: String) async throws { + let model = FirebaseAI.componentInstance(config).generativeModel( + modelName: modelName, + generationConfig: GenerationConfig( + responseModalities: [.audio], + speechConfig: SpeechConfig(voiceName: "Kore") + ) + ) + let prompt = TextPart( + "Welcome aboard flight 412 with direct service to Tokyo Haneda.", + speechMetadata: SpeechMetadata(style: "professional and welcoming") + ) + + let response = try await model.generateContent(prompt) + + let candidate = try #require(response.candidates.first) + #expect(candidate.finishReason == .stop) + let audioPart = try #require(candidate.content.parts.first as? InlineDataPart) + #expect(!audioPart.data.isEmpty) + #expect(audioPart.mimeType == unaryMIMEType) + } + + @Test(arguments: InstanceConfig.defaultConfigs, ttsModels) + func singleSpeakerStream_withTurnMetadata(_ config: InstanceConfig, + modelName: String) async throws { + let model = FirebaseAI.componentInstance(config).generativeModel( + modelName: modelName, + generationConfig: GenerationConfig( + responseModalities: [.audio], + speechConfig: SpeechConfig(voiceName: "Puck") + ) + ) + let prompt = TextPart( + "Deep in the ancient forest, a glowing stone lay hidden beneath the roots of the world tree.", + speechMetadata: SpeechMetadata(style: "whispering and mysterious") + ) + + let responseStream = try model.generateContentStream(prompt) + + var receivedAudioDataCount = 0 + for try await chunk in responseStream { + guard let candidate = chunk.candidates.first else { continue } + if let audioPart = candidate.content.parts.first as? InlineDataPart { + #expect(!audioPart.data.isEmpty) + #expect(audioPart.mimeType == streamingMIMEType) + receivedAudioDataCount += audioPart.data.count + } else { + #expect(candidate.finishReason == .stop, "Unexpected non-audio chunk: \(candidate)") + } + } + #expect(receivedAudioDataCount > 0) + } + + @Test(arguments: InstanceConfig.defaultConfigs, ttsModels) + func singleSpeaker_withoutTurnMetadata(_ config: InstanceConfig, modelName: String) async throws { + let model = FirebaseAI.componentInstance(config).generativeModel( + modelName: modelName, + generationConfig: GenerationConfig( + responseModalities: [.audio], + speechConfig: SpeechConfig(voiceName: "Charon", languageCode: "fr-CA") + ) + ) + + let response = try await model.generateContent("Attache ta tuque!") + + let candidate = try #require(response.candidates.first) + #expect(candidate.finishReason == .stop) + let audioPart = try #require(candidate.content.parts.first as? InlineDataPart) + #expect(!audioPart.data.isEmpty) + #expect(audioPart.mimeType == unaryMIMEType) + } + + @Test(arguments: InstanceConfig.defaultConfigs, ttsModels) + func singleSpeakerStream_withoutTurnMetadata(_ config: InstanceConfig, + modelName: String) async throws { + let model = FirebaseAI.componentInstance(config).generativeModel( + modelName: modelName, + generationConfig: GenerationConfig( + responseModalities: [.audio], + speechConfig: SpeechConfig(voiceName: "Charon", languageCode: "en-GB") + ) + ) + + let responseStream = try model.generateContentStream(""" + Breaking news: The James Webb Space Telescope has detected signs of water vapour on a distant \ + exoplanet. + """) + + var receivedAudioDataCount = 0 + for try await chunk in responseStream { + guard let candidate = chunk.candidates.first else { continue } + if let audioPart = candidate.content.parts.first as? InlineDataPart { + #expect(!audioPart.data.isEmpty) + #expect(audioPart.mimeType == streamingMIMEType) + receivedAudioDataCount += audioPart.data.count + } else { + #expect(candidate.finishReason == .stop, "Unexpected non-audio chunk: \(candidate)") + } + } + #expect(receivedAudioDataCount > 0) + } + + // MARK: - Multi-Speaker Tests + + @Test(arguments: InstanceConfig.defaultConfigs, ttsModels) + func multiSpeaker_withTurnMetadata(_ config: InstanceConfig, modelName: String) async throws { + let model = FirebaseAI.componentInstance(config).generativeModel( + modelName: modelName, + generationConfig: GenerationConfig( + responseModalities: [.audio], + speechConfig: SpeechConfig( + multiSpeakerVoiceConfig: MultiSpeakerVoiceConfig( + speakerVoiceConfigs: [ + SpeakerVoiceConfig(speaker: "Joe", voiceName: "Puck"), + SpeakerVoiceConfig(speaker: "Jane", voiceName: "Kore"), + ] + ) + ) + ) + ) + let turn1 = TextPart( + "Did you see the northern lights last night? The entire sky turned brilliant emerald green!", + speechMetadata: SpeechMetadata(speaker: "Joe", style: "awed and enthusiastic") + ) + let turn2 = TextPart( + "I did! It was breathtaking. I've never seen anything quite like it.", + speechMetadata: SpeechMetadata(speaker: "Jane", style: "wonderstruck and peaceful") + ) + + let response = try await model.generateContent(turn1, turn2) + + let candidate = try #require(response.candidates.first) + #expect(candidate.finishReason == .stop) + let audioPart = try #require(candidate.content.parts.first as? InlineDataPart) + #expect(!audioPart.data.isEmpty) + #expect(audioPart.mimeType == unaryMIMEType) + } + + @Test(arguments: InstanceConfig.defaultConfigs, ttsModels) + func multiSpeakerStream_withTurnMetadata(_ config: InstanceConfig, + modelName: String) async throws { + let model = FirebaseAI.componentInstance(config).generativeModel( + modelName: modelName, + generationConfig: GenerationConfig( + responseModalities: [.audio], + speechConfig: SpeechConfig( + multiSpeakerVoiceConfig: MultiSpeakerVoiceConfig( + speakerVoiceConfigs: [ + SpeakerVoiceConfig(speaker: "Joe", voiceName: "Puck"), + SpeakerVoiceConfig(speaker: "Jane", voiceName: "Kore"), + ] + ) + ) + ) + ) + let turn1 = TextPart( + "Welcome back everyone. Today we're debating: is time travel theoretically possible?", + speechMetadata: SpeechMetadata(speaker: "Joe", style: "inquisitive and energetic") + ) + let turn2 = TextPart( + "According to general relativity, travelling forward is easy; going backward, not so much.", + speechMetadata: SpeechMetadata(speaker: "Jane", style: "analytical and wry") + ) + + let responseStream = try model.generateContentStream(turn1, turn2) + + var receivedAudioDataCount = 0 + for try await chunk in responseStream { + guard let candidate = chunk.candidates.first else { continue } + if let audioPart = candidate.content.parts.first as? InlineDataPart { + #expect(!audioPart.data.isEmpty) + #expect(audioPart.mimeType == streamingMIMEType) + receivedAudioDataCount += audioPart.data.count + } else { + #expect(candidate.finishReason == .stop, "Unexpected non-audio chunk: \(candidate)") + } + } + #expect(receivedAudioDataCount > 0) + } + + // MARK: - Error Handling Tests + + @Test(arguments: InstanceConfig.defaultConfigs, ttsModels) + func multiSpeaker_invalidSize(_ config: InstanceConfig, modelName: String) async throws { + let model = FirebaseAI.componentInstance(config).generativeModel( + modelName: modelName, + generationConfig: GenerationConfig( + responseModalities: [.audio], + speechConfig: SpeechConfig( + multiSpeakerVoiceConfig: MultiSpeakerVoiceConfig( + speakerVoiceConfigs: [ + SpeakerVoiceConfig(speaker: "Speaker1", voiceName: "Puck"), + SpeakerVoiceConfig(speaker: "Speaker2", voiceName: "Charon"), + SpeakerVoiceConfig(speaker: "Speaker3", voiceName: "Aoede"), + ] + ) + ) + ) + ) + + let turn1 = TextPart("Hello!", speechMetadata: SpeechMetadata(speaker: "Speaker1")) + let turn2 = TextPart("Hi!", speechMetadata: SpeechMetadata(speaker: "Speaker2")) + let turn3 = TextPart("Hey!", speechMetadata: SpeechMetadata(speaker: "Speaker3")) + let error = try await #require(throws: GenerateContentError.self) { + try await model.generateContent(turn1, turn2, turn3) + } + guard case let .internalError(underlyingError) = error else { + Issue.record("Expected internalError; got \(error).") + return + } + #expect( + String(describing: underlyingError) + .contains("the number of speaker_voice_configs must equal 2") + ) + } +} diff --git a/FirebaseAI/Tests/Unit/ChatTests.swift b/FirebaseAI/Tests/Unit/ChatTests.swift index 6d4e5b80658..4441624cde5 100644 --- a/FirebaseAI/Tests/Unit/ChatTests.swift +++ b/FirebaseAI/Tests/Unit/ChatTests.swift @@ -196,6 +196,67 @@ final class ChatTests: XCTestCase { XCTAssertEqual(chat.history, history) } + func testAggregatedChunks_speechMetadata_notCombined() throws { + let history = History(history: []) + let chunk1 = ModelContent(role: "model", parts: [TextPart("Regular text A")]) + let chunk2 = ModelContent( + role: "model", + parts: [ + TextPart("Spoken turn", speechMetadata: SpeechMetadata(speaker: "Joe", style: "cheerful")), + ] + ) + let chunk3 = ModelContent(role: "model", parts: [TextPart("Regular text B")]) + let chunk4 = ModelContent(role: "model", parts: [TextPart(" and C")]) + + let result = history.aggregatedChunks([chunk1, chunk2, chunk3, chunk4]) + + XCTAssertEqual(result.parts.count, 3) + let part1 = try XCTUnwrap(result.parts[0] as? TextPart) + XCTAssertEqual(part1.text, "Regular text A") + XCTAssertNil(part1.speechMetadata) + let part2 = try XCTUnwrap(result.parts[1] as? TextPart) + XCTAssertEqual(part2.text, "Spoken turn") + XCTAssertEqual(part2.speechMetadata, SpeechMetadata(speaker: "Joe", style: "cheerful")) + let part3 = try XCTUnwrap(result.parts[2] as? TextPart) + XCTAssertEqual(part3.text, "Regular text B and C") + XCTAssertNil(part3.speechMetadata) + } + + func testAggregatedChunks_adjacentSpeechMetadata_notCombined() throws { + let history = History(history: []) + let chunk1 = ModelContent( + role: "model", + parts: [ + TextPart("First turn", speechMetadata: SpeechMetadata(speaker: "Joe", style: "cheerful")), + ] + ) + let chunk2 = ModelContent( + role: "model", + parts: [ + TextPart("Second turn", speechMetadata: SpeechMetadata(speaker: "Joe", style: "cheerful")), + ] + ) + let chunk3 = ModelContent( + role: "model", + parts: [ + TextPart("Third turn", speechMetadata: SpeechMetadata(speaker: "Jane", style: "calm")), + ] + ) + + let result = history.aggregatedChunks([chunk1, chunk2, chunk3]) + + XCTAssertEqual(result.parts.count, 3) + let part1 = try XCTUnwrap(result.parts[0] as? TextPart) + XCTAssertEqual(part1.text, "First turn") + XCTAssertEqual(part1.speechMetadata, SpeechMetadata(speaker: "Joe", style: "cheerful")) + let part2 = try XCTUnwrap(result.parts[1] as? TextPart) + XCTAssertEqual(part2.text, "Second turn") + XCTAssertEqual(part2.speechMetadata, SpeechMetadata(speaker: "Joe", style: "cheerful")) + let part3 = try XCTUnwrap(result.parts[2] as? TextPart) + XCTAssertEqual(part3.text, "Third turn") + XCTAssertEqual(part3.speechMetadata, SpeechMetadata(speaker: "Jane", style: "calm")) + } + func testSendMessage_unary_codeExecution_appendsHistory() async throws { MockURLProtocol.requestHandler = try GenerativeModelTestUtil.httpRequestHandler( forResource: "unary-success-code-execution", diff --git a/FirebaseAI/Tests/Unit/PartTests.swift b/FirebaseAI/Tests/Unit/PartTests.swift index 9839de06325..cb044263989 100644 --- a/FirebaseAI/Tests/Unit/PartTests.swift +++ b/FirebaseAI/Tests/Unit/PartTests.swift @@ -41,6 +41,26 @@ final class PartTests: XCTestCase { let part = try decoder.decode(TextPart.self, from: jsonData) XCTAssertEqual(part.text, expectedText) + XCTAssertNil(part.speechMetadata) + } + + func testDecodeTextPart_withSpeechMetadata() throws { + let expectedText = "Hello, world!" + let json = """ + { + "speechMetadata" : { + "speaker" : "Joe", + "style" : "cheerful" + }, + "text" : "\(expectedText)" + } + """ + let jsonData = try XCTUnwrap(json.data(using: .utf8)) + + let part = try decoder.decode(TextPart.self, from: jsonData) + + XCTAssertEqual(part.text, expectedText) + XCTAssertEqual(part.speechMetadata, SpeechMetadata(speaker: "Joe", style: "cheerful")) } func testDecodeInlineDataPart() throws { @@ -227,6 +247,25 @@ final class PartTests: XCTestCase { """) } + func testEncodeTextPart_withSpeechMetadata() throws { + let expectedText = "Hello, world!" + let speechMetadata = SpeechMetadata(speaker: "Joe", style: "cheerful") + let textPart = TextPart(expectedText, speechMetadata: speechMetadata) + + let jsonData = try encoder.encode(textPart) + + let json = try XCTUnwrap(String(data: jsonData, encoding: .utf8)) + XCTAssertEqual(json, """ + { + "speechMetadata" : { + "speaker" : "Joe", + "style" : "cheerful" + }, + "text" : "\(expectedText)" + } + """) + } + func testEncodeInlineDataPart() throws { let mimeType = "image/png" let imageBase64 = try PartTests.blueSquareImage() diff --git a/FirebaseAI/Tests/Unit/Types/InternalPartTests.swift b/FirebaseAI/Tests/Unit/Types/InternalPartTests.swift index 92cf0113808..5663a426671 100644 --- a/FirebaseAI/Tests/Unit/Types/InternalPartTests.swift +++ b/FirebaseAI/Tests/Unit/Types/InternalPartTests.swift @@ -17,6 +17,11 @@ import XCTest final class InternalPartTests: XCTestCase { let decoder = JSONDecoder() + let encoder = JSONEncoder() + + override func setUp() { + encoder.outputFormatting = [.prettyPrinted, .sortedKeys, .withoutEscapingSlashes] + } func testDecodeTextPartWithThought() throws { let json = """ @@ -495,6 +500,140 @@ final class InternalPartTests: XCTestCase { XCTAssertNil(codeExecutionResult.output) } + func testDecodeTextPartWithSpeechMetadata() throws { + let json = """ + { + "text": "Have a wonderful day!", + "speechMetadata": { + "speaker": "Joe", + "style": "cheerful and friendly" + } + } + """ + let jsonData = try XCTUnwrap(json.data(using: .utf8)) + + let part = try decoder.decode(InternalPart.self, from: jsonData) + + XCTAssertNil(part.isThought) + guard case let .text(text) = part.data else { + XCTFail("Decoded part is not a text part.") + return + } + XCTAssertEqual(text, "Have a wonderful day!") + XCTAssertEqual( + part.speechMetadata, + SpeechMetadata(speaker: "Joe", style: "cheerful and friendly") + ) + } + + func testDecodeTextPartWithSpeechMetadata_speakerOnly() throws { + let json = """ + { + "text": "How's it going today Jane?", + "speechMetadata": { + "speaker": "Joe" + } + } + """ + let jsonData = try XCTUnwrap(json.data(using: .utf8)) + + let part = try decoder.decode(InternalPart.self, from: jsonData) + + XCTAssertNil(part.isThought) + guard case let .text(text) = part.data else { + XCTFail("Decoded part is not a text part.") + return + } + XCTAssertEqual(text, "How's it going today Jane?") + XCTAssertEqual(part.speechMetadata, SpeechMetadata(speaker: "Joe")) + } + + func testDecodeTextPartWithSpeechMetadata_styleOnly() throws { + let json = """ + { + "text": "Have a wonderful day!", + "speechMetadata": { + "style": "cheerful and friendly" + } + } + """ + let jsonData = try XCTUnwrap(json.data(using: .utf8)) + + let part = try decoder.decode(InternalPart.self, from: jsonData) + + XCTAssertNil(part.isThought) + guard case let .text(text) = part.data else { + XCTFail("Decoded part is not a text part.") + return + } + XCTAssertEqual(text, "Have a wonderful day!") + XCTAssertEqual(part.speechMetadata, SpeechMetadata(style: "cheerful and friendly")) + } + + func testEncodeTextPartWithSpeechMetadata() throws { + let part = InternalPart( + .text("Have a wonderful day!"), + speechMetadata: SpeechMetadata(speaker: "Joe", style: "cheerful and friendly"), + isThought: nil, + thoughtSignature: nil + ) + + let jsonData = try encoder.encode(part) + + let json = try XCTUnwrap(String(data: jsonData, encoding: .utf8)) + XCTAssertEqual(json, """ + { + "speechMetadata" : { + "speaker" : "Joe", + "style" : "cheerful and friendly" + }, + "text" : "Have a wonderful day!" + } + """) + } + + func testEncodeTextPartWithSpeechMetadata_speakerOnly() throws { + let part = InternalPart( + .text("How's it going today Jane?"), + speechMetadata: SpeechMetadata(speaker: "Joe"), + isThought: nil, + thoughtSignature: nil + ) + + let jsonData = try encoder.encode(part) + + let json = try XCTUnwrap(String(data: jsonData, encoding: .utf8)) + XCTAssertEqual(json, """ + { + "speechMetadata" : { + "speaker" : "Joe" + }, + "text" : "How's it going today Jane?" + } + """) + } + + func testEncodeTextPartWithSpeechMetadata_styleOnly() throws { + let part = InternalPart( + .text("Have a wonderful day!"), + speechMetadata: SpeechMetadata(style: "cheerful and friendly"), + isThought: nil, + thoughtSignature: nil + ) + + let jsonData = try encoder.encode(part) + + let json = try XCTUnwrap(String(data: jsonData, encoding: .utf8)) + XCTAssertEqual(json, """ + { + "speechMetadata" : { + "style" : "cheerful and friendly" + }, + "text" : "Have a wonderful day!" + } + """) + } + func testEncodeUnsupportedPart_doesNotCrash() throws { let json = """ {