audio

package
v0.20.0 Latest Latest
Warning

This package is not in the latest version of its module.

Go to latest
Published: Sep 15, 2026 License: AGPL-3.0 Imports: 36 Imported by: 0

Documentation

Index

Constants

This section is empty.

Variables

This section is empty.

Functions

func SyncRegistry

func SyncRegistry(ctx context.Context, logger *slog.Logger, queries dbstore.Queries, registry *Registry) error

Types

type AudioConfig

type AudioConfig struct {
	Format     string      `json:"format"`
	SampleRate int         `json:"sample_rate"`
	Speed      float64     `json:"speed"`
	Pitch      float64     `json:"pitch"`
	Voice      VoiceConfig `json:"voice"`
}

AudioConfig is kept for backward compatibility with the legacy Edge adapter tests.

func (AudioConfig) Validate

func (AudioConfig) Validate() error

type ConfigSchema

type ConfigSchema struct {
	Fields []FieldSchema `json:"fields"`
}

type FieldSchema

type FieldSchema struct {
	Key         string   `json:"key"`
	Type        string   `json:"type"`
	Title       string   `json:"title,omitempty"`
	Description string   `json:"description,omitempty"`
	Required    bool     `json:"required,omitempty"`
	Advanced    bool     `json:"advanced,omitempty"`
	Enum        []string `json:"enum,omitempty"`
	Example     any      `json:"example,omitempty"`
	Order       int      `json:"order"`
}

FieldSchema describes a single dynamic speech config field.

type ImportModelsResponse

type ImportModelsResponse struct {
	Created int      `json:"created"`
	Skipped int      `json:"skipped"`
	Models  []string `json:"models"`
}

ImportModelsResponse represents the response for importing speech models.

type ModelCapabilities

type ModelCapabilities struct {
	ConfigSchema ConfigSchema      `json:"config_schema,omitempty"`
	Voices       []VoiceInfo       `json:"voices,omitempty"`
	Formats      []string          `json:"formats,omitempty"`
	Speed        *ParamConstraint  `json:"speed,omitempty"`
	Pitch        *ParamConstraint  `json:"pitch,omitempty"`
	Metadata     map[string]string `json:"metadata,omitempty"`
}

ModelCapabilities exposes optional UX hints for speech config forms.

type ModelInfo

type ModelInfo struct {
	ID           string            `json:"id"`
	Name         string            `json:"name"`
	Description  string            `json:"description,omitempty"`
	TemplateOnly bool              `json:"template_only,omitempty"`
	ConfigSchema ConfigSchema      `json:"config_schema,omitempty"`
	Capabilities ModelCapabilities `json:"capabilities"`
}

ModelInfo describes a single speech model exposed by a provider definition.

type ParamConstraint

type ParamConstraint struct {
	Options []float64 `json:"options,omitempty"`
	Min     float64   `json:"min,omitempty"`
	Max     float64   `json:"max,omitempty"`
	Default float64   `json:"default"`
}

ParamConstraint describes valid values for a numeric parameter. If Options is non-empty, only those discrete values are allowed.

type ProviderDefinition

type ProviderDefinition struct {
	ClientType                models.ClientType
	DisplayName               string
	Icon                      string
	Description               string
	ConfigSchema              ConfigSchema
	DefaultModel              string
	SupportsList              bool
	Models                    []ModelInfo
	Factory                   ProviderFactory
	DefaultTranscriptionModel string
	SupportsTranscriptionList bool
	TranscriptionModels       []ModelInfo
	TranscriptionFactory      TranscriptionProviderFactory
	Order                     int
}

type ProviderFactory

type ProviderFactory func(config map[string]any) (sdk.SpeechProvider, error)

type ProviderMetaResponse

type ProviderMetaResponse struct {
	Provider                  string       `json:"provider"`
	DisplayName               string       `json:"display_name"`
	Description               string       `json:"description"`
	ConfigSchema              ConfigSchema `json:"config_schema,omitempty"`
	DefaultModel              string       `json:"default_model,omitempty"`
	Models                    []ModelInfo  `json:"models,omitempty"`
	DefaultSynthesisModel     string       `json:"default_synthesis_model,omitempty"`
	SynthesisModels           []ModelInfo  `json:"synthesis_models,omitempty"`
	SupportsSynthesisList     bool         `json:"supports_synthesis_list,omitempty"`
	DefaultTranscriptionModel string       `json:"default_transcription_model,omitempty"`
	TranscriptionModels       []ModelInfo  `json:"transcription_models,omitempty"`
	SupportsTranscriptionList bool         `json:"supports_transcription_list,omitempty"`
}

ProviderMetaResponse exposes adapter metadata (from the registry, not DB).

type Registry

type Registry struct {
	// contains filtered or unexported fields
}

func NewRegistry

func NewRegistry() *Registry

func (*Registry) Get

func (r *Registry) Get(clientType models.ClientType) (ProviderDefinition, error)

func (*Registry) List

func (r *Registry) List() []ProviderDefinition

func (*Registry) ListMeta

func (r *Registry) ListMeta() []ProviderMetaResponse

func (*Registry) ListSpeechMeta

func (r *Registry) ListSpeechMeta() []ProviderMetaResponse

func (*Registry) ListTranscriptionMeta

func (r *Registry) ListTranscriptionMeta() []ProviderMetaResponse

func (*Registry) Register

func (r *Registry) Register(def ProviderDefinition)

type Service

type Service struct {
	// contains filtered or unexported fields
}

func NewService

func NewService(log *slog.Logger, queries dbstore.Queries, registry *Registry) *Service

func (*Service) FetchRemoteModels

func (s *Service) FetchRemoteModels(ctx context.Context, providerID string) ([]ModelInfo, error)

func (*Service) FetchRemoteTranscriptionModels

func (s *Service) FetchRemoteTranscriptionModels(ctx context.Context, providerID string) ([]ModelInfo, error)

func (*Service) GetModelCapabilities

func (s *Service) GetModelCapabilities(ctx context.Context, modelID string) (*ModelCapabilities, error)

func (*Service) GetSpeechModel

func (s *Service) GetSpeechModel(ctx context.Context, id string) (SpeechModelResponse, error)

func (*Service) GetSpeechModelCapabilities

func (s *Service) GetSpeechModelCapabilities(ctx context.Context, modelID string) (*ModelCapabilities, error)

func (*Service) GetSpeechProvider

func (s *Service) GetSpeechProvider(ctx context.Context, id string) (SpeechProviderResponse, error)

func (*Service) GetTranscriptionModel

func (s *Service) GetTranscriptionModel(ctx context.Context, id string) (TranscriptionModelResponse, error)

func (*Service) GetTranscriptionModelCapabilities

func (s *Service) GetTranscriptionModelCapabilities(ctx context.Context, modelID string) (*ModelCapabilities, error)

func (*Service) ListMeta

func (s *Service) ListMeta(_ context.Context) []ProviderMetaResponse

func (*Service) ListSpeechMeta

func (s *Service) ListSpeechMeta(_ context.Context) []ProviderMetaResponse

func (*Service) ListSpeechModels

func (s *Service) ListSpeechModels(ctx context.Context) ([]SpeechModelResponse, error)

func (*Service) ListSpeechModelsByProvider

func (s *Service) ListSpeechModelsByProvider(ctx context.Context, providerID string) ([]SpeechModelResponse, error)

func (*Service) ListSpeechProviders

func (s *Service) ListSpeechProviders(ctx context.Context) ([]SpeechProviderResponse, error)

func (*Service) ListTranscriptionMeta

func (s *Service) ListTranscriptionMeta(_ context.Context) []ProviderMetaResponse

func (*Service) ListTranscriptionModels

func (s *Service) ListTranscriptionModels(ctx context.Context) ([]TranscriptionModelResponse, error)

func (*Service) ListTranscriptionModelsByProvider

func (s *Service) ListTranscriptionModelsByProvider(ctx context.Context, providerID string) ([]TranscriptionModelResponse, error)

func (*Service) ListTranscriptionProviders

func (s *Service) ListTranscriptionProviders(ctx context.Context) ([]SpeechProviderResponse, error)

func (*Service) Registry

func (s *Service) Registry() *Registry

func (*Service) StreamToFile

func (s *Service) StreamToFile(ctx context.Context, modelID string, text string, w io.Writer) (string, error)

func (*Service) Synthesize

func (s *Service) Synthesize(ctx context.Context, modelID string, text string, overrideCfg map[string]any) ([]byte, string, error)

func (*Service) Transcribe

func (s *Service) Transcribe(ctx context.Context, modelID string, audio []byte, filename string, contentType string, overrideCfg map[string]any) (*sdk.TranscriptionResult, error)

func (*Service) UpdateSpeechModel

func (s *Service) UpdateSpeechModel(ctx context.Context, id string, req UpdateSpeechModelRequest) (SpeechModelResponse, error)

func (*Service) UpdateTranscriptionModel

func (s *Service) UpdateTranscriptionModel(ctx context.Context, id string, req UpdateSpeechModelRequest) (TranscriptionModelResponse, error)

type SpeechModelResponse

type SpeechModelResponse struct {
	ID           string         `json:"id"`
	ModelID      string         `json:"model_id"`
	Name         string         `json:"name"`
	ProviderID   string         `json:"provider_id"`
	ProviderType string         `json:"provider_type,omitempty"`
	Config       map[string]any `json:"config,omitempty"`
	CreatedAt    time.Time      `json:"created_at"`
	UpdatedAt    time.Time      `json:"updated_at"`
}

SpeechModelResponse represents a speech model from the unified models table.

type SpeechProviderResponse

type SpeechProviderResponse struct {
	ID         string         `json:"id"`
	Name       string         `json:"name"`
	ClientType string         `json:"client_type"`
	Icon       string         `json:"icon,omitempty"`
	Enable     bool           `json:"enable"`
	Config     map[string]any `json:"config,omitempty"`
	CreatedAt  time.Time      `json:"created_at"`
	UpdatedAt  time.Time      `json:"updated_at"`
}

SpeechProviderResponse represents a speech-capable provider from the unified providers table.

type TempStore

type TempStore struct {
	// contains filtered or unexported fields
}

TempStore manages temporary audio files on disk with automatic TTL-based cleanup. MIME type and other metadata are NOT stored here — they travel in the tool result JSON through the SSE stream.

func NewTempStore

func NewTempStore(baseDir string) (*TempStore, error)

NewTempStore creates a TempStore under the given base directory.

func (*TempStore) Create

func (s *TempStore) Create() (id string, f *os.File, err error)

Create opens a new temporary file for writing. The caller writes audio data into the returned file and must close it when done.

func (*TempStore) Delete

func (s *TempStore) Delete(id string)

Delete removes a temp file and its tracking entry.

func (*TempStore) FileSize

func (s *TempStore) FileSize(id string) (int64, error)

FileSize returns the size of the temp file in bytes.

func (*TempStore) ReadAndDelete

func (s *TempStore) ReadAndDelete(id string) ([]byte, error)

ReadAndDelete reads the full file contents and removes the entry.

func (*TempStore) StartCleanup

func (s *TempStore) StartCleanup(done <-chan struct{})

StartCleanup runs a background goroutine that removes expired entries.

type TestSynthesizeRequest

type TestSynthesizeRequest struct {
	Text   string         `json:"text"`
	Config map[string]any `json:"config,omitempty"`
}

TestSynthesizeRequest represents a text-to-speech test request.

type TestTranscriptionRequest

type TestTranscriptionRequest struct {
	Config map[string]any `json:"config,omitempty"`
}

TestTranscriptionRequest represents an audio-to-text test request.

type TestTranscriptionResponse

type TestTranscriptionResponse struct {
	Text            string              `json:"text"`
	Language        string              `json:"language,omitempty"`
	DurationSeconds float64             `json:"duration_seconds,omitempty"`
	Words           []TranscriptionWord `json:"words,omitempty"`
	Metadata        map[string]any      `json:"metadata,omitempty"`
}

TestTranscriptionResponse represents the result of a transcription test.

type TranscriptionModelResponse

type TranscriptionModelResponse struct {
	ID           string         `json:"id"`
	ModelID      string         `json:"model_id"`
	Name         string         `json:"name"`
	ProviderID   string         `json:"provider_id"`
	ProviderType string         `json:"provider_type,omitempty"`
	Config       map[string]any `json:"config,omitempty"`
	CreatedAt    time.Time      `json:"created_at"`
	UpdatedAt    time.Time      `json:"updated_at"`
}

TranscriptionModelResponse represents a transcription model from the unified models table.

type TranscriptionProviderFactory

type TranscriptionProviderFactory func(config map[string]any) (sdk.TranscriptionProvider, error)

type TranscriptionWord

type TranscriptionWord struct {
	Text      string  `json:"text"`
	Start     float64 `json:"start,omitempty"`
	End       float64 `json:"end,omitempty"`
	SpeakerID string  `json:"speaker_id,omitempty"`
}

TranscriptionWord represents a single word alignment from a transcription result.

type TtsAdapter

type TtsAdapter interface {
	Type() TtsType
	Meta() TtsMeta
	DefaultModel() string
	Models() []ModelInfo
	ResolveModel(model string) (string, error)
	Synthesize(ctx context.Context, text string, model string, config AudioConfig) ([]byte, error)
	Stream(ctx context.Context, text string, model string, config AudioConfig) (chan []byte, chan error)
}

type TtsMeta

type TtsMeta struct {
	Provider    string
	Description string
}

type TtsType

type TtsType string

type UpdateSpeechModelRequest

type UpdateSpeechModelRequest struct {
	Name   *string        `json:"name,omitempty"`
	Config map[string]any `json:"config,omitempty"`
}

UpdateSpeechModelRequest is used for updating a speech model.

type UpdateSpeechProviderRequest

type UpdateSpeechProviderRequest struct {
	Name   *string `json:"name,omitempty"`
	Enable *bool   `json:"enable,omitempty"`
}

UpdateSpeechProviderRequest is used for updating a speech provider.

type VoiceConfig

type VoiceConfig struct {
	ID   string `json:"id"`
	Lang string `json:"lang"`
}

VoiceConfig is kept for backward compatibility with the legacy Edge adapter tests.

type VoiceInfo

type VoiceInfo struct {
	ID   string `json:"id"`
	Name string `json:"name"`
	Lang string `json:"lang"`
}

Directories

Path Synopsis
adapter
alibabacloud
Package alibabacloud implements DashScope's OpenAI-compatible Qwen ASR API.
Package alibabacloud implements DashScope's OpenAI-compatible Qwen ASR API.

Jump to

Keyboard shortcuts

? : This menu
/ : Search site
f or F : Jump to
y or Y : Canonical URL