realtime

package
v0.13.0 Latest Latest
Warning

This package is not in the latest version of its module.

Go to latest
Published: Jun 21, 2026 License: MIT Imports: 8 Imported by: 0

Documentation

Overview

Package realtime provides WebSocket-based real-time TTS and STT services.

This package provides low-latency streaming capabilities for text-to-speech and speech-to-text using WebSocket connections.

Example:

import (
    "github.com/plexusone/elevenlabs-go"
    "github.com/plexusone/elevenlabs-go/realtime"
)

client, _ := elevenlabs.NewClient()
conn, _ := client.Realtime().ConnectTTS(ctx, "voice_id", &realtime.TTSOptions{...})
conn.SendText("Hello, world!")

Index

Constants

This section is empty.

Variables

This section is empty.

Functions

This section is empty.

Types

type STTConnection

type STTConnection struct {
	// contains filtered or unexported fields
}

STTConnection represents an active WebSocket STT connection.

func (*STTConnection) Close

func (wsc *STTConnection) Close() error

Close closes the WebSocket connection gracefully.

func (*STTConnection) Commit

func (wsc *STTConnection) Commit() error

Commit forces a commit of the current transcript segment.

func (*STTConnection) Errors

func (wsc *STTConnection) Errors() <-chan error

Errors returns a channel that receives errors.

func (*STTConnection) SendAudio

func (wsc *STTConnection) SendAudio(audio []byte) error

SendAudio sends audio data for transcription.

func (*STTConnection) SessionID

func (wsc *STTConnection) SessionID() string

SessionID returns the session ID assigned by the server.

func (*STTConnection) StreamAudio

func (wsc *STTConnection) StreamAudio(ctx context.Context, audioStream <-chan []byte) (<-chan *STTTranscript, <-chan error)

StreamAudio is a convenience method that streams audio from a channel. It handles committing automatically when the input channel closes.

func (*STTConnection) Transcripts

func (wsc *STTConnection) Transcripts() <-chan *STTTranscript

Transcripts returns a channel that receives transcription results.

type STTOptions

type STTOptions struct {
	// ModelID is the transcription model to use.
	ModelID string

	// AudioFormat specifies the audio encoding format.
	AudioFormat string

	// LanguageCode is the expected language.
	LanguageCode string

	// IncludeTimestamps enables word-level timing information.
	IncludeTimestamps bool

	// IncludeLanguageDetection includes detected language in responses.
	IncludeLanguageDetection bool

	// CommitStrategy determines how transcripts are committed.
	CommitStrategy string

	// VAD settings
	VADSilenceThresholdSecs float64
	VADThreshold            float64
	MinSpeechDurationMs     int
	MinSilenceDurationMs    int
}

STTOptions configures the WebSocket STT connection.

func DefaultSTTOptions

func DefaultSTTOptions() *STTOptions

DefaultSTTOptions returns default options for real-time STT.

type STTTranscript

type STTTranscript struct {
	Text         string    `json:"text"`
	IsFinal      bool      `json:"is_final"`
	Words        []STTWord `json:"words,omitempty"`
	LanguageCode string    `json:"language_code,omitempty"`
}

STTTranscript represents a transcription result.

type STTWord

type STTWord struct {
	Text      string  `json:"text"`
	Start     float64 `json:"start"`
	End       float64 `json:"end"`
	Type      string  `json:"type,omitempty"`
	SpeakerID string  `json:"speaker_id,omitempty"`
}

STTWord represents a single word with timing.

type Service

type Service struct {
	// contains filtered or unexported fields
}

Service handles real-time WebSocket TTS and STT.

func New

func New(apiKey, baseURL string) *Service

New creates a new realtime service.

func (*Service) ConnectSTT

func (s *Service) ConnectSTT(ctx context.Context, opts *STTOptions) (*STTConnection, error)

ConnectSTT establishes a WebSocket connection for real-time STT.

func (*Service) ConnectTTS

func (s *Service) ConnectTTS(ctx context.Context, voiceID string, opts *TTSOptions) (*TTSConnection, error)

ConnectTTS establishes a WebSocket connection for real-time TTS.

type TTSAlignment

type TTSAlignment struct {
	Characters     []string  `json:"characters"`
	CharacterStart []float64 `json:"character_start_times_seconds"`
	CharacterEnd   []float64 `json:"character_end_times_seconds"`
}

TTSAlignment contains word-level timing information.

type TTSConnection

type TTSConnection struct {
	// contains filtered or unexported fields
}

TTSConnection represents an active WebSocket TTS connection.

func (*TTSConnection) Alignments

func (wsc *TTSConnection) Alignments() <-chan *TTSAlignment

Alignments returns a channel that receives word alignment information.

func (*TTSConnection) Audio

func (wsc *TTSConnection) Audio() <-chan []byte

Audio returns a channel that receives audio chunks.

func (*TTSConnection) Close

func (wsc *TTSConnection) Close() error

Close closes the WebSocket connection gracefully.

func (*TTSConnection) Done

func (wsc *TTSConnection) Done() <-chan struct{}

Done returns a channel that is closed when all audio has been received after Flush().

func (*TTSConnection) Errors

func (wsc *TTSConnection) Errors() <-chan error

Errors returns a channel that receives errors.

func (*TTSConnection) Flush

func (wsc *TTSConnection) Flush() error

Flush signals that no more text will be sent and flushes remaining audio.

func (*TTSConnection) SendText

func (wsc *TTSConnection) SendText(text string) error

SendText sends text to be converted to speech.

func (*TTSConnection) StreamText

func (wsc *TTSConnection) StreamText(ctx context.Context, textStream <-chan string) (<-chan []byte, <-chan error)

StreamText is a convenience method that sends all text from a channel and returns audio. It handles flushing automatically when the input channel closes.

func (*TTSConnection) TriggerGeneration

func (wsc *TTSConnection) TriggerGeneration() error

TriggerGeneration forces audio generation for buffered text.

type TTSOptions

type TTSOptions struct {
	// ModelID is the model to use. Defaults to "eleven_turbo_v2_5" for low latency.
	ModelID string

	// OutputFormat specifies the audio output format.
	// Recommended for real-time: "pcm_16000", "pcm_22050", "pcm_24000", "pcm_44100"
	OutputFormat string

	// VoiceSettings configures the voice parameters.
	VoiceSettings *VoiceSettings

	// OptimizeStreamingLatency reduces latency at the cost of quality (0-4).
	OptimizeStreamingLatency int

	// EnableSSMLParsing enables SSML parsing for the input text.
	EnableSSMLParsing bool

	// LanguageCode is the ISO language code (e.g., "en", "es").
	LanguageCode string

	// ChunkLengthSchedule controls text chunking for audio generation.
	ChunkLengthSchedule []int

	// InactivityTimeout is the context timeout in seconds (default 20).
	InactivityTimeout int

	// PronunciationDictionaryIDs is a list of pronunciation dictionary IDs to use.
	PronunciationDictionaryIDs []string
}

TTSOptions configures the WebSocket TTS connection.

func DefaultTTSOptions

func DefaultTTSOptions() *TTSOptions

DefaultTTSOptions returns default options optimized for low latency.

type VoiceSettings

type VoiceSettings struct {
	Stability       float64
	SimilarityBoost float64
	Style           float64
	Speed           float64
	UseSpeakerBoost bool
}

VoiceSettings contains voice configuration for TTS.

func DefaultVoiceSettings

func DefaultVoiceSettings() *VoiceSettings

DefaultVoiceSettings returns sensible default voice settings for real-time TTS.

Jump to

Keyboard shortcuts

? : This menu
/ : Search site
f or F : Jump to
y or Y : Canonical URL