Documentation
¶
Overview ¶
Package realtime provides WebSocket-based real-time TTS and STT services.
This package provides low-latency streaming capabilities for text-to-speech and speech-to-text using WebSocket connections.
Example:
import (
"github.com/plexusone/elevenlabs-go"
"github.com/plexusone/elevenlabs-go/realtime"
)
client, _ := elevenlabs.NewClient()
conn, _ := client.Realtime().ConnectTTS(ctx, "voice_id", &realtime.TTSOptions{...})
conn.SendText("Hello, world!")
Index ¶
- type STTConnection
- func (wsc *STTConnection) Close() error
- func (wsc *STTConnection) Commit() error
- func (wsc *STTConnection) Errors() <-chan error
- func (wsc *STTConnection) SendAudio(audio []byte) error
- func (wsc *STTConnection) SessionID() string
- func (wsc *STTConnection) StreamAudio(ctx context.Context, audioStream <-chan []byte) (<-chan *STTTranscript, <-chan error)
- func (wsc *STTConnection) Transcripts() <-chan *STTTranscript
- type STTOptions
- type STTTranscript
- type STTWord
- type Service
- type TTSAlignment
- type TTSConnection
- func (wsc *TTSConnection) Alignments() <-chan *TTSAlignment
- func (wsc *TTSConnection) Audio() <-chan []byte
- func (wsc *TTSConnection) Close() error
- func (wsc *TTSConnection) Done() <-chan struct{}
- func (wsc *TTSConnection) Errors() <-chan error
- func (wsc *TTSConnection) Flush() error
- func (wsc *TTSConnection) SendText(text string) error
- func (wsc *TTSConnection) StreamText(ctx context.Context, textStream <-chan string) (<-chan []byte, <-chan error)
- func (wsc *TTSConnection) TriggerGeneration() error
- type TTSOptions
- type VoiceSettings
Constants ¶
This section is empty.
Variables ¶
This section is empty.
Functions ¶
This section is empty.
Types ¶
type STTConnection ¶
type STTConnection struct {
// contains filtered or unexported fields
}
STTConnection represents an active WebSocket STT connection.
func (*STTConnection) Close ¶
func (wsc *STTConnection) Close() error
Close closes the WebSocket connection gracefully.
func (*STTConnection) Commit ¶
func (wsc *STTConnection) Commit() error
Commit forces a commit of the current transcript segment.
func (*STTConnection) Errors ¶
func (wsc *STTConnection) Errors() <-chan error
Errors returns a channel that receives errors.
func (*STTConnection) SendAudio ¶
func (wsc *STTConnection) SendAudio(audio []byte) error
SendAudio sends audio data for transcription.
func (*STTConnection) SessionID ¶
func (wsc *STTConnection) SessionID() string
SessionID returns the session ID assigned by the server.
func (*STTConnection) StreamAudio ¶
func (wsc *STTConnection) StreamAudio(ctx context.Context, audioStream <-chan []byte) (<-chan *STTTranscript, <-chan error)
StreamAudio is a convenience method that streams audio from a channel. It handles committing automatically when the input channel closes.
func (*STTConnection) Transcripts ¶
func (wsc *STTConnection) Transcripts() <-chan *STTTranscript
Transcripts returns a channel that receives transcription results.
type STTOptions ¶
type STTOptions struct {
// ModelID is the transcription model to use.
ModelID string
// AudioFormat specifies the audio encoding format.
AudioFormat string
// LanguageCode is the expected language.
LanguageCode string
// IncludeTimestamps enables word-level timing information.
IncludeTimestamps bool
// IncludeLanguageDetection includes detected language in responses.
IncludeLanguageDetection bool
// CommitStrategy determines how transcripts are committed.
CommitStrategy string
// VAD settings
VADSilenceThresholdSecs float64
VADThreshold float64
MinSpeechDurationMs int
MinSilenceDurationMs int
}
STTOptions configures the WebSocket STT connection.
func DefaultSTTOptions ¶
func DefaultSTTOptions() *STTOptions
DefaultSTTOptions returns default options for real-time STT.
type STTTranscript ¶
type STTTranscript struct {
Text string `json:"text"`
IsFinal bool `json:"is_final"`
Words []STTWord `json:"words,omitempty"`
LanguageCode string `json:"language_code,omitempty"`
}
STTTranscript represents a transcription result.
type STTWord ¶
type STTWord struct {
Text string `json:"text"`
Start float64 `json:"start"`
End float64 `json:"end"`
Type string `json:"type,omitempty"`
SpeakerID string `json:"speaker_id,omitempty"`
}
STTWord represents a single word with timing.
type Service ¶
type Service struct {
// contains filtered or unexported fields
}
Service handles real-time WebSocket TTS and STT.
func (*Service) ConnectSTT ¶
func (s *Service) ConnectSTT(ctx context.Context, opts *STTOptions) (*STTConnection, error)
ConnectSTT establishes a WebSocket connection for real-time STT.
func (*Service) ConnectTTS ¶
func (s *Service) ConnectTTS(ctx context.Context, voiceID string, opts *TTSOptions) (*TTSConnection, error)
ConnectTTS establishes a WebSocket connection for real-time TTS.
type TTSAlignment ¶
type TTSAlignment struct {
Characters []string `json:"characters"`
CharacterStart []float64 `json:"character_start_times_seconds"`
CharacterEnd []float64 `json:"character_end_times_seconds"`
}
TTSAlignment contains word-level timing information.
type TTSConnection ¶
type TTSConnection struct {
// contains filtered or unexported fields
}
TTSConnection represents an active WebSocket TTS connection.
func (*TTSConnection) Alignments ¶
func (wsc *TTSConnection) Alignments() <-chan *TTSAlignment
Alignments returns a channel that receives word alignment information.
func (*TTSConnection) Audio ¶
func (wsc *TTSConnection) Audio() <-chan []byte
Audio returns a channel that receives audio chunks.
func (*TTSConnection) Close ¶
func (wsc *TTSConnection) Close() error
Close closes the WebSocket connection gracefully.
func (*TTSConnection) Done ¶
func (wsc *TTSConnection) Done() <-chan struct{}
Done returns a channel that is closed when all audio has been received after Flush().
func (*TTSConnection) Errors ¶
func (wsc *TTSConnection) Errors() <-chan error
Errors returns a channel that receives errors.
func (*TTSConnection) Flush ¶
func (wsc *TTSConnection) Flush() error
Flush signals that no more text will be sent and flushes remaining audio.
func (*TTSConnection) SendText ¶
func (wsc *TTSConnection) SendText(text string) error
SendText sends text to be converted to speech.
func (*TTSConnection) StreamText ¶
func (wsc *TTSConnection) StreamText(ctx context.Context, textStream <-chan string) (<-chan []byte, <-chan error)
StreamText is a convenience method that sends all text from a channel and returns audio. It handles flushing automatically when the input channel closes.
func (*TTSConnection) TriggerGeneration ¶
func (wsc *TTSConnection) TriggerGeneration() error
TriggerGeneration forces audio generation for buffered text.
type TTSOptions ¶
type TTSOptions struct {
// ModelID is the model to use. Defaults to "eleven_turbo_v2_5" for low latency.
ModelID string
// OutputFormat specifies the audio output format.
// Recommended for real-time: "pcm_16000", "pcm_22050", "pcm_24000", "pcm_44100"
OutputFormat string
// VoiceSettings configures the voice parameters.
VoiceSettings *VoiceSettings
// OptimizeStreamingLatency reduces latency at the cost of quality (0-4).
OptimizeStreamingLatency int
// EnableSSMLParsing enables SSML parsing for the input text.
EnableSSMLParsing bool
// LanguageCode is the ISO language code (e.g., "en", "es").
LanguageCode string
// ChunkLengthSchedule controls text chunking for audio generation.
ChunkLengthSchedule []int
// InactivityTimeout is the context timeout in seconds (default 20).
InactivityTimeout int
// PronunciationDictionaryIDs is a list of pronunciation dictionary IDs to use.
PronunciationDictionaryIDs []string
}
TTSOptions configures the WebSocket TTS connection.
func DefaultTTSOptions ¶
func DefaultTTSOptions() *TTSOptions
DefaultTTSOptions returns default options optimized for low latency.
type VoiceSettings ¶
type VoiceSettings struct {
Stability float64
SimilarityBoost float64
Style float64
Speed float64
UseSpeakerBoost bool
}
VoiceSettings contains voice configuration for TTS.
func DefaultVoiceSettings ¶
func DefaultVoiceSettings() *VoiceSettings
DefaultVoiceSettings returns sensible default voice settings for real-time TTS.