Skip to main content

SpeechModel Interface

Interface that all speech synthesis (text-to-speech) model implementations must satisfy.

Interface Definition​

type SpeechModel interface {
// Metadata
SpecificationVersion() string
Provider() string
ModelID() string

// Speech synthesis
DoGenerate(ctx context.Context, opts *SpeechGenerateOptions) (*types.SpeechResult, error)
}

Methods​

MethodParametersReturnsDescription
SpecificationVersion()-stringSpecification version
Provider()-stringProvider name
ModelID()-stringModel identifier
DoGenerate()ctx, *SpeechGenerateOptions*types.SpeechResult, errorGenerate speech audio

SpeechGenerateOptions​

type SpeechGenerateOptions struct {
Text string
Voice string
Speed *float64
}

Examples​

Using a Speech Model​

package main

import (
"context"
"log"
"os"

"github.com/digitallysavvy/go-ai/pkg/provider"
"github.com/digitallysavvy/go-ai/pkg/providers/openai"
)

func main() {
p := openai.New(openai.Config{
APIKey: "your-api-key",
})
model, err := p.SpeechModel("tts-1")
if err != nil {
log.Fatal(err)
}

result, err := model.DoGenerate(context.Background(), &provider.SpeechGenerateOptions{
Text: "Hello, world!",
Voice: "alloy",
})
if err != nil {
log.Fatal(err)
}

os.WriteFile("output.mp3", result.Audio, 0644)
}

See Also​