Skip to content
5 changes: 5 additions & 0 deletions FirebaseAI/CHANGELOG.md
Original file line number Diff line number Diff line change
@@ -1,4 +1,9 @@
# Unreleased
- [feature] **Public Preview**: Added `SpeechMetadata` support to `TextPart`
for Gemini text-to-speech (TTS) models, enabling turn-level speaker routing
in multi-speaker synthesis and sustained speech delivery styling. See the
[speech generation guide](https://firebase.google.com/docs/ai-logic/generate-speech)
for more details.
- [fixed] Fixed crashes when a chat session's history contained parts with
unrecognized data or code execution parts received from the server, when
creating a `ModelContent` with an unsupported `Part` type, and when decoding a
Expand Down
7 changes: 5 additions & 2 deletions FirebaseAI/Sources/History.swift
Original file line number Diff line number Diff line change
Expand Up @@ -63,7 +63,9 @@ final class History: Sendable {
// Loop through all the parts, aggregating the text.
for part in chunks.flatMap({ $0.internalParts }) {
// Only text parts may be combined.
if case let .text(text) = part.data, part.thoughtSignature == nil {
if case let .text(text) = part.data,
part.thoughtSignature == nil,
part.speechMetadata == nil {
// Thought summaries must not be combined with regular text.
if part.isThought ?? false {
// If we were combining regular text, flush it before handling "thoughts".
Expand All @@ -79,7 +81,8 @@ final class History: Sendable {
combinedText += text
}
} else {
// This is a non-combinable part (not text), flush any pending text.
// This is a non-combinable part (non-text, signed thought, or speech metadata), flush
// any pending text.
flush()
parts.append(part)
}
Expand Down
19 changes: 17 additions & 2 deletions FirebaseAI/Sources/ModelContent.swift
Original file line number Diff line number Diff line change
Expand Up @@ -58,8 +58,14 @@ struct InternalPart: Equatable, Sendable {

let thoughtSignature: String?

init(_ data: OneOfData, isThought: Bool?, thoughtSignature: String?) {
let speechMetadata: SpeechMetadata?

init(_ data: OneOfData,
speechMetadata: SpeechMetadata? = nil,
isThought: Bool?,
thoughtSignature: String?) {
self.data = data
self.speechMetadata = speechMetadata
self.isThought = isThought
self.thoughtSignature = thoughtSignature
}
Expand All @@ -78,7 +84,12 @@ public struct ModelContent: Equatable, Sendable {
return internalParts.compactMap { part -> (any Part)? in
switch part.data {
case let .text(text):
return TextPart(text, isThought: part.isThought, thoughtSignature: part.thoughtSignature)
return TextPart(
text,
speechMetadata: part.speechMetadata,
isThought: part.isThought,
thoughtSignature: part.thoughtSignature
)
case let .inlineData(inlineData):
return InlineDataPart(
inlineData, isThought: part.isThought, thoughtSignature: part.thoughtSignature
Expand Down Expand Up @@ -126,6 +137,7 @@ public struct ModelContent: Equatable, Sendable {
case let textPart as TextPart:
convertedParts.append(InternalPart(
.text(textPart.text),
speechMetadata: textPart.speechMetadata,
isThought: textPart._isThought,
thoughtSignature: textPart.thoughtSignature
))
Expand Down Expand Up @@ -215,6 +227,7 @@ extension InternalPart: Codable {
enum CodingKeys: String, CodingKey {
case isThought = "thought"
case thoughtSignature
case speechMetadata
}

public func encode(to encoder: Encoder) throws {
Expand All @@ -227,6 +240,7 @@ extension InternalPart: Codable {
var container = encoder.container(keyedBy: CodingKeys.self)
try container.encodeIfPresent(isThought, forKey: .isThought)
try container.encodeIfPresent(thoughtSignature, forKey: .thoughtSignature)
try container.encodeIfPresent(speechMetadata, forKey: .speechMetadata)
}

public init(from decoder: Decoder) throws {
Expand All @@ -241,6 +255,7 @@ extension InternalPart: Codable {
let container = try decoder.container(keyedBy: CodingKeys.self)
isThought = try container.decodeIfPresent(Bool.self, forKey: .isThought)
thoughtSignature = try container.decodeIfPresent(String.self, forKey: .thoughtSignature)
speechMetadata = try container.decodeIfPresent(SpeechMetadata.self, forKey: .speechMetadata)
}
}

Expand Down
19 changes: 13 additions & 6 deletions FirebaseAI/Sources/Types/Public/MultiSpeakerVoiceConfig.swift
Original file line number Diff line number Diff line change
Expand Up @@ -14,23 +14,30 @@

import Foundation

/// Configuration for a multi-speaker audio generation setup.
/// **[Public Preview]** Configuration for a multi-speaker audio generation setup.
///
/// **Public Preview**: This API is a public preview and may be subject to change.
/// > Warning: This API is a public preview and may be subject to change.
///
/// Enables the model to generate audio containing multiple distinct speakers, alternating voices
/// dynamically based on speaker labels in the prompt.
/// dynamically based on the `speaker` specified in each turn's ``SpeechMetadata``.
///
/// > Warning: Multi-speaker configurations are not currently supported by the Live API (e.g.,
/// > `LiveGenerationConfig`).
/// > Important: When using multi-speaker generation, every ``TextPart`` in the request prompt must
/// > include ``SpeechMetadata`` with a `speaker` matching one of the configured speakers. Omitting
/// > `speaker` in a multi-speaker request results in a backend error.
///
/// > Warning: Multi-speaker configurations are not currently supported by the Live API, such as
/// > ``LiveGenerationConfig``.
///
/// For more details, see the
/// [multi-speaker guide](https://firebase.google.com/docs/ai-logic/generate-speech#multi-speaker).
public struct MultiSpeakerVoiceConfig: Sendable {
let multiSpeakerVoiceConfig: ProtoMultiSpeakerVoiceConfig

init(_ multiSpeakerVoiceConfig: ProtoMultiSpeakerVoiceConfig) {
self.multiSpeakerVoiceConfig = multiSpeakerVoiceConfig
}

/// Creates a configuration for the multi-speaker setup.
/// Creates a multi-speaker voice configuration.
///
/// - Parameters:
/// - speakerVoiceConfigs: A list of voice configurations for the participating speakers.
Expand Down
35 changes: 33 additions & 2 deletions FirebaseAI/Sources/Types/Public/Part.swift
Original file line number Diff line number Diff line change
Expand Up @@ -31,18 +31,49 @@ public struct TextPart: Part {
/// Text value.
public let text: String

/// **[Public Preview]** Optional speech metadata configuring the speaker and delivery style for
/// speech synthesis.
///
/// When using a Gemini text-to-speech model, this metadata controls turn-level delivery style
/// and speaker assignment for multi-speaker audio generation.
///
/// > Important: When using a multi-speaker configuration, `speechMetadata` with a matching
/// > `speaker` is required on every ``TextPart``. Omitting `speaker` in a multi-speaker request
/// > results in a backend error.
///
/// For more details, see the
/// [Text-to-speech guide](https://firebase.google.com/docs/ai-logic/generate-speech).
public let speechMetadata: SpeechMetadata?

public var isThought: Bool { _isThought ?? false }

let thoughtSignature: String?

let _isThought: Bool?

/// Creates a text part with a string value.
///
/// - Parameter text: The text string. For speech generation, this represents the verbatim
/// transcript to be synthesized.
public init(_ text: String) {
self.init(text, isThought: nil, thoughtSignature: nil)
self.init(text, speechMetadata: nil, isThought: nil, thoughtSignature: nil)
}

/// Creates a text part with a string value and optional speech metadata.
///
/// - Parameters:
/// - text: The text string. For speech generation, this represents the verbatim transcript
/// to be synthesized.
/// - speechMetadata: Optional ``SpeechMetadata`` configuring the speaker and delivery style for
/// text-to-speech generation. Defaults to `nil`.
public init(_ text: String, speechMetadata: SpeechMetadata? = nil) {
self.init(text, speechMetadata: speechMetadata, isThought: nil, thoughtSignature: nil)
}

init(_ text: String, isThought: Bool?, thoughtSignature: String?) {
init(_ text: String, speechMetadata: SpeechMetadata? = nil, isThought: Bool?,
thoughtSignature: String?) {
self.text = text
self.speechMetadata = speechMetadata
_isThought = isThought
self.thoughtSignature = thoughtSignature
}
Expand Down
15 changes: 10 additions & 5 deletions FirebaseAI/Sources/Types/Public/SpeakerVoiceConfig.swift
Original file line number Diff line number Diff line change
Expand Up @@ -14,9 +14,13 @@

import Foundation

/// Configures a speaker with a unique name/identifier and a specific voice.
/// **[Public Preview]** Configuration pairing a speaker name or identifier with a preset voice.
///
/// **Public Preview**: This API is a public preview and may be subject to change.
/// > Warning: This API is a public preview and may be subject to change.
///
/// This configuration pairs a speaker name (such as `"Alice"` or `"Joe"`) with a preset voice name.
/// The `speaker` name defined here must match the `speaker` string specified in ``SpeechMetadata``
/// on each dialogue turn's ``TextPart`` in a multi-speaker request.
public struct SpeakerVoiceConfig: Sendable {
let speakerVoiceConfig: ProtoSpeakerVoiceConfig

Expand All @@ -27,11 +31,12 @@ public struct SpeakerVoiceConfig: Sendable {
/// Creates a configuration for a speaker using a voice name.
///
/// - Parameters:
/// - speaker: The unique name/identifier of the speaker (e.g., `"Alice"`).
/// - speaker: The unique name or identifier of the speaker (for example, `"Alice"`). This name
/// must be passed to ``SpeechMetadata/init(speaker:style:)`` for this speaker's turns.
/// - voiceName: The name of the preset voice to assign to this speaker.
///
/// Find the list of supported voices at
/// https://firebase.google.com/docs/ai-logic/generate-speech#supported-voices-and-languages
/// For a list of available voices, see the documentation on
/// [voices](https://firebase.google.com/docs/ai-logic/generate-speech#response-voices).
public init(speaker: String, voiceName: String) {
self.init(
ProtoSpeakerVoiceConfig(
Expand Down
38 changes: 28 additions & 10 deletions FirebaseAI/Sources/Types/Public/SpeechConfig.swift
Original file line number Diff line number Diff line change
Expand Up @@ -14,12 +14,18 @@

import Foundation

/// Speech configuration class for controlling the model's speech and audio generation behaviors.
/// **[Public Preview]** Configuration for model speech and audio generation behaviors.
///
/// **Public Preview**: This API is a public preview and may be subject to change.
/// > Warning: This API is a public preview and may be subject to change.
///
/// This allows you to configure the voice properties (single-speaker OR multi-speaker setup) and
/// language preferences when requesting the model to generate spoken responses.
/// Configures voice properties (single-speaker or multi-speaker setup) and language preferences
/// when requesting the model to generate spoken responses.
///
/// For turn-level speech delivery control (such as emotion, pacing, and whispering), attach
/// ``SpeechMetadata`` to individual ``TextPart`` instances in your request prompt.
///
/// For more details on speech generation, see the
/// [Text-to-speech guide](https://firebase.google.com/docs/ai-logic/generate-speech).
public struct SpeechConfig: Sendable {
let speechConfig: ProtoSpeechConfig

Expand All @@ -32,13 +38,13 @@ public struct SpeechConfig: Sendable {
/// - Parameters:
/// - voiceName: The name of the prebuilt voice to be used for the model's speech response.
///
/// To learn more about the available voices, see the docs on
/// [Voice options](https://ai.google.dev/gemini-api/docs/speech-generation#voices)\.
/// For available voices, see the documentation on
/// [voices](https://firebase.google.com/docs/ai-logic/generate-speech#response-voices).
/// - languageCode: BCP-47 language code to use when parsing text sent from the client, instead
/// of audio. By default, the model will attempt to detect the input language automatically.
///
/// To learn which codes are supported, see the docs on
/// [Supported languages](https://ai.google.dev/gemini-api/docs/speech-generation#languages)\.
/// For supported language codes, see the documentation on
/// [languages](https://firebase.google.com/docs/ai-logic/generate-speech#languages).
public init(voiceName: String, languageCode: String? = nil) {
self.init(
ProtoSpeechConfig(
Expand All @@ -50,13 +56,25 @@ public struct SpeechConfig: Sendable {

/// Creates a new ``SpeechConfig`` value for a multi-speaker setup.
///
/// > Warning: Multi-speaker configurations are not currently supported by the Live API (e.g.,
/// > `LiveGenerationConfig`).
/// > Warning: Multi-speaker configurations are not currently supported by the Live API, such as
/// > ``LiveGenerationConfig``.
///
/// - Parameters:
/// - multiSpeakerVoiceConfig: The configuration detailing multiple speakers and their
/// corresponding voices.
///
/// > Important: When using a multi-speaker configuration, each dialogue turn in the
/// > request prompt must be passed as a separate ``TextPart`` with ``SpeechMetadata``
/// > specifying a `speaker` matching one of the configured speakers. `speaker` is required
/// > on every part in a multi-speaker request.
///
/// See the documentation on
/// [multi-speaker](https://firebase.google.com/docs/ai-logic/generate-speech#multi-speaker)
/// for more details.
/// - languageCode: BCP-47 language code to use when parsing text sent from the client.
///
/// For supported language codes, see the documentation on
/// [languages](https://firebase.google.com/docs/ai-logic/generate-speech#languages).
public init(multiSpeakerVoiceConfig: MultiSpeakerVoiceConfig, languageCode: String? = nil) {
self.init(
ProtoSpeechConfig(
Expand Down
71 changes: 71 additions & 0 deletions FirebaseAI/Sources/Types/Public/SpeechMetadata.swift
Original file line number Diff line number Diff line change
@@ -0,0 +1,71 @@
// Copyright 2026 Google LLC
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.

/// **[Public Preview]** Turn-level speech synthesis metadata for text content in a model request.
///
/// > Warning: This API is a public preview and may be subject to change.
///
/// When generating audio using Gemini text-to-speech (TTS) models, the input text in a
/// ``TextPart`` is treated strictly as a verbatim transcript. ``SpeechMetadata`` attaches
/// structured turn-level instructions to control speaker assignment and speech delivery.
///
/// For more details on speech generation, see the
/// [Text-to-speech guide](https://firebase.google.com/docs/ai-logic/generate-speech).
public struct SpeechMetadata: Sendable, Equatable, Hashable {
/// The unique name or identifier of the speaker for multi-speaker synthesis.
///
/// > Important: When using a multi-speaker configuration, `speaker` is required on every
/// > ``TextPart``. Omitting `speaker` in a multi-speaker request results in a backend error.
let speaker: String?

/// The sustained delivery style instruction for speech synthesis.
let style: String?

/// Creates speech metadata with an optional speaker identifier and delivery style.
///
/// - Parameters:
/// - speaker: The unique name or identifier of the speaker for multi-speaker synthesis
/// (for example, `"Joe"`). Must match a speaker name configured in ``SpeakerVoiceConfig``.
///
/// > Important: When using a multi-speaker configuration, `speaker` is required on every
/// > dialogue turn. Pass each speaker turn as a separate ``TextPart`` with a matching
/// > `speaker` name.
///
/// See the documentation on
/// [multi-speaker](https://firebase.google.com/docs/ai-logic/generate-speech#multi-speaker)
/// for more details. Defaults to `nil`.
/// - style: A natural-language description of the sustained delivery style, emotion,
/// prosody, pacing, or volume across the entire turn (for example,
/// `"cheerful and friendly"` or `"whispering urgently"`).
///
/// Instructions should be partitioned based on scope:
/// - **Sustained turn-level delivery (`style`)**: Put attributes that apply across an
/// entire dialogue turn into `style` (for example, `"whispering"`,
/// `"cheerful and friendly"`, `"speaking slowly"`, `"out of breath"`, or `"sarcastic"`).
/// - **Point-in-time events (inline vocal tags)**: Place momentary non-speech vocalizations
/// or pauses directly inside the transcript text using angle brackets (for example,
/// `<laugh>`, `<sigh>`, `<cough>`, `<breath>`, or `<short pause>`), rather than in
/// `style`.
///
/// See [audio tags](https://firebase.google.com/docs/ai-logic/generate-speech#audio-tags)
/// for more details. Defaults to `nil`.
public init(speaker: String? = nil, style: String? = nil) {
self.speaker = speaker
self.style = style
}
}

// MARK: - Codable Conformance

extension SpeechMetadata: Codable {}
3 changes: 2 additions & 1 deletion FirebaseAI/Tests/TestApp/Sources/Constants.swift
Original file line number Diff line number Diff line change
Expand Up @@ -29,6 +29,7 @@ public enum ModelNames {
public static let gemini2_5_Pro = "gemini-2.5-pro"
public static let gemini3_1_FlashLite = "gemini-3.1-flash-lite"
public static let gemini3_1_FlashImage = "gemini-3.1-flash-image"
public static let gemini3_1_FlashTTSPreview = "gemini-3.1-flash-tts-preview"
public static let gemini3_8_FlashTTS = "gemini-3.8-flash-tts"
public static let gemini3_8_FlashLiteTTS = "gemini-3.8-flash-lite-tts"
public static let gemma4_31B = "gemma-4-31b-it"
}
Loading
Loading