Documentation
¶
Overview ¶
Speech recognition with Next-gen Kaldi.
sherpa-onnx is an open-source speech recognition framework for Next-gen Kaldi. It depends only on onnxruntime, supporting both streaming and non-streaming speech recognition.
It does not need to access the network during recognition and everything runs locally.
It supports a variety of platforms, such as Linux (x86_64, aarch64, arm), Windows (x86_64, x86), macOS (x86_64, arm64), etc.
Usage examples:
Real-time speech recognition from a microphone
Decode files using a non-streaming model
Please see https://github.com/k2-fsa/sherpa-onnx/tree/master/go-api-examples/non-streaming-decode-files
Decode files using a streaming model
Please see https://github.com/k2-fsa/sherpa-onnx/tree/master/go-api-examples/streaming-decode-files
Convert text to speech using a non-streaming model
Please see https://github.com/k2-fsa/sherpa-onnx/tree/master/go-api-examples/non-streaming-tts
Index ¶
- func DeleteAudioTagging(tagging *AudioTagging)
- func DeleteCircularBuffer(buffer *CircularBuffer)
- func DeleteKeywordSpotter(spotter *KeywordSpotter)
- func DeleteOfflinePunc(punc *OfflinePunctuation)
- func DeleteOfflineRecognizer(recognizer *OfflineRecognizer)
- func DeleteOfflineSpeakerDiarization(sd *OfflineSpeakerDiarization)
- func DeleteOfflineSpeechDenoiser(sd *OfflineSpeechDenoiser)
- func DeleteOfflineStream(stream *OfflineStream)
- func DeleteOfflineTts(tts *OfflineTts)
- func DeleteOnlinePunctuation(punc *OnlinePunctuation)
- func DeleteOnlineRecognizer(recognizer *OnlineRecognizer)
- func DeleteOnlineSpeechDenoiser(sd *OnlineSpeechDenoiser)
- func DeleteOnlineStream(stream *OnlineStream)
- func DeleteSpeakerEmbeddingExtractor(ex *SpeakerEmbeddingExtractor)
- func DeleteSpeakerEmbeddingManager(m *SpeakerEmbeddingManager)
- func DeleteSpokenLanguageIdentification(slid *SpokenLanguageIdentification)
- func DeleteVoiceActivityDetector(vad *VoiceActivityDetector)
- func GetGitDate() string
- func GetGitSha1() string
- func GetVersion() string
- type AudioBuffer
- type AudioEvent
- type AudioTagging
- type AudioTaggingConfig
- type AudioTaggingModelConfig
- type CircularBuffer
- type DenoisedAudio
- type FastClusteringConfig
- type FeatureConfig
- type GeneratedAudio
- type GenerationConfig
- type HomophoneReplacerConfig
- type KeywordSpotter
- type KeywordSpotterConfig
- type KeywordSpotterResult
- type OfflineCanaryModelConfig
- type OfflineCohereTranscribeModelConfig
- type OfflineDolphinModelConfig
- type OfflineFireRedAsrCtcModelConfig
- type OfflineFireRedAsrModelConfig
- type OfflineFunASRNanoModelConfig
- type OfflineLMConfig
- type OfflineMedAsrCtcModelConfig
- type OfflineModelConfig
- type OfflineMoonshineModelConfig
- type OfflineNemoEncDecCtcModelConfig
- type OfflineOmnilingualAsrCtcModelConfig
- type OfflineParaformerModelConfig
- type OfflinePunctuation
- type OfflinePunctuationConfig
- type OfflinePunctuationModelConfig
- type OfflineQwen3ASRModelConfig
- type OfflineRecognizer
- type OfflineRecognizerConfig
- type OfflineRecognizerResult
- type OfflineSenseVoiceModelConfig
- type OfflineSourceSeparationConfig
- type OfflineSourceSeparationModelConfig
- type OfflineSourceSeparationSpleeterModelConfig
- type OfflineSourceSeparationUvrModelConfig
- type OfflineSpeakerDiarization
- func (sd *OfflineSpeakerDiarization) Process(samples []float32) []OfflineSpeakerDiarizationSegment
- func (sd *OfflineSpeakerDiarization) ProcessWithContext(ctx context.Context, samples []float32, ...) ([]OfflineSpeakerDiarizationSegment, error)
- func (sd *OfflineSpeakerDiarization) ProcessWithProgressCallback(samples []float32, cb OfflineSpeakerDiarizationProgressCallback) []OfflineSpeakerDiarizationSegment
- func (sd *OfflineSpeakerDiarization) SampleRate() int
- func (sd *OfflineSpeakerDiarization) SetConfig(config *OfflineSpeakerDiarizationConfig)
- type OfflineSpeakerDiarizationConfig
- type OfflineSpeakerDiarizationProgressCallback
- type OfflineSpeakerDiarizationSegment
- type OfflineSpeakerSegmentationModelConfig
- type OfflineSpeakerSegmentationPyannoteModelConfig
- type OfflineSpeechDenoiser
- type OfflineSpeechDenoiserConfig
- type OfflineSpeechDenoiserDpdfNetModelConfig
- type OfflineSpeechDenoiserGtcrnModelConfig
- type OfflineSpeechDenoiserModelConfig
- type OfflineStream
- type OfflineTdnnModelConfig
- type OfflineTransducerModelConfig
- type OfflineTts
- func (tts *OfflineTts) Generate(text string, sid int, speed float32) *GeneratedAudio
- func (tts *OfflineTts) GenerateWithCallback(text string, sid int, speed float32, ...) *GeneratedAudio
- func (tts *OfflineTts) GenerateWithConfig(text string, cfg *GenerationConfig, ...) *GeneratedAudio
- func (tts *OfflineTts) GenerateWithProgressCallback(text string, sid int, speed float32, ...) *GeneratedAudio
- func (tts *OfflineTts) GenerateWithZipvoice(text, promptText string, promptSamples []float32, promptSampleRate int, ...) *GeneratedAudiodeprecated
- func (tts *OfflineTts) NumSpeakers() int
- func (tts *OfflineTts) SampleRate() int
- type OfflineTtsConfig
- type OfflineTtsKittenModelConfig
- type OfflineTtsKokoroModelConfig
- type OfflineTtsMatchaModelConfig
- type OfflineTtsModelConfig
- type OfflineTtsPocketModelConfig
- type OfflineTtsSupertonicModelConfig
- type OfflineTtsVitsModelConfig
- type OfflineTtsZipvoiceModelConfig
- type OfflineWenetCtcModelConfig
- type OfflineWhisperModelConfig
- type OfflineZipformerAudioTaggingModelConfig
- type OfflineZipformerCtcModelConfig
- type OnlineCtcFstDecoderConfig
- type OnlineModelConfig
- type OnlineNemoCtcModelConfig
- type OnlineParaformerModelConfig
- type OnlinePunctuation
- type OnlinePunctuationConfig
- type OnlinePunctuationModelConfig
- type OnlineRecognizer
- func (recognizer *OnlineRecognizer) Decode(s *OnlineStream)
- func (recognizer *OnlineRecognizer) DecodeStreams(s []*OnlineStream)
- func (recognizer *OnlineRecognizer) GetResult(s *OnlineStream) *OnlineRecognizerResult
- func (recognizer *OnlineRecognizer) IsEndpoint(s *OnlineStream) bool
- func (recognizer *OnlineRecognizer) IsReady(s *OnlineStream) bool
- func (recognizer *OnlineRecognizer) Reset(s *OnlineStream)
- type OnlineRecognizerConfig
- type OnlineRecognizerResult
- type OnlineSpeechDenoiser
- type OnlineSpeechDenoiserConfig
- type OnlineStream
- type OnlineToneCtcModelConfig
- type OnlineTransducerModelConfig
- type OnlineZipformer2CtcModelConfig
- type SileroVadModelConfig
- type SourceSeparator
- type SpeakerEmbeddingExtractor
- type SpeakerEmbeddingExtractorConfig
- type SpeakerEmbeddingManager
- func (m *SpeakerEmbeddingManager) AllSpeakers() []string
- func (m *SpeakerEmbeddingManager) Contains(name string) bool
- func (m *SpeakerEmbeddingManager) NumSpeakers() int
- func (m *SpeakerEmbeddingManager) Register(name string, embedding []float32) bool
- func (m *SpeakerEmbeddingManager) RegisterV(name string, embeddings [][]float32) bool
- func (m *SpeakerEmbeddingManager) Remove(name string) bool
- func (m *SpeakerEmbeddingManager) Search(embedding []float32, threshold float32) string
- func (m *SpeakerEmbeddingManager) Verify(name string, embedding []float32, threshold float32) bool
- type SpeechSegment
- type SpokenLanguageIdentification
- type SpokenLanguageIdentificationConfig
- type SpokenLanguageIdentificationResult
- type SpokenLanguageIdentificationWhisperConfig
- type TenVadModelConfig
- type VadModelConfig
- type VoiceActivityDetector
- func (vad *VoiceActivityDetector) AcceptWaveform(samples []float32)
- func (vad *VoiceActivityDetector) Clear()
- func (vad *VoiceActivityDetector) Flush()
- func (vad *VoiceActivityDetector) Front() *SpeechSegment
- func (vad *VoiceActivityDetector) IsEmpty() bool
- func (vad *VoiceActivityDetector) IsSpeech() bool
- func (vad *VoiceActivityDetector) Pop()
- func (vad *VoiceActivityDetector) Reset()
- type Wave
Constants ¶
This section is empty.
Variables ¶
This section is empty.
Functions ¶
func DeleteAudioTagging ¶
func DeleteAudioTagging(tagging *AudioTagging)
func DeleteCircularBuffer ¶
func DeleteCircularBuffer(buffer *CircularBuffer)
func DeleteKeywordSpotter ¶
func DeleteKeywordSpotter(spotter *KeywordSpotter)
Free the internal pointer inside the recognizer to avoid memory leak.
func DeleteOfflinePunc ¶
func DeleteOfflinePunc(punc *OfflinePunctuation)
func DeleteOfflineRecognizer ¶
func DeleteOfflineRecognizer(recognizer *OfflineRecognizer)
Frees the internal pointer of the recognition to avoid memory leak.
func DeleteOfflineSpeakerDiarization ¶
func DeleteOfflineSpeakerDiarization(sd *OfflineSpeakerDiarization)
func DeleteOfflineSpeechDenoiser ¶
func DeleteOfflineSpeechDenoiser(sd *OfflineSpeechDenoiser)
Free the internal pointer inside the OfflineSpeechDenoiser to avoid memory leak.
func DeleteOfflineStream ¶
func DeleteOfflineStream(stream *OfflineStream)
Frees the internal pointer of the stream to avoid memory leak.
func DeleteOfflineTts ¶
func DeleteOfflineTts(tts *OfflineTts)
Free the internal pointer inside the tts to avoid memory leak.
func DeleteOnlinePunctuation ¶
func DeleteOnlinePunctuation(punc *OnlinePunctuation)
func DeleteOnlineRecognizer ¶
func DeleteOnlineRecognizer(recognizer *OnlineRecognizer)
Free the internal pointer inside the recognizer to avoid memory leak.
func DeleteOnlineSpeechDenoiser ¶
func DeleteOnlineSpeechDenoiser(sd *OnlineSpeechDenoiser)
Free the internal pointer inside the OnlineSpeechDenoiser to avoid memory leak.
func DeleteOnlineStream ¶
func DeleteOnlineStream(stream *OnlineStream)
Delete the internal pointer inside the stream to avoid memory leak.
func DeleteSpeakerEmbeddingExtractor ¶
func DeleteSpeakerEmbeddingExtractor(ex *SpeakerEmbeddingExtractor)
func DeleteSpeakerEmbeddingManager ¶
func DeleteSpeakerEmbeddingManager(m *SpeakerEmbeddingManager)
func DeleteSpokenLanguageIdentification ¶
func DeleteSpokenLanguageIdentification(slid *SpokenLanguageIdentification)
func DeleteVoiceActivityDetector ¶
func DeleteVoiceActivityDetector(vad *VoiceActivityDetector)
func GetGitDate ¶
func GetGitDate() string
func GetGitSha1 ¶
func GetGitSha1() string
func GetVersion ¶
func GetVersion() string
Types ¶
type AudioBuffer ¶
type AudioBuffer struct {
Samples []float32
ChannelCount int
SampleRate int
SamplesPerChannel int
// contains filtered or unexported fields
}
func NewAudioBuffer ¶
func NewAudioBuffer(samples []float32, channelCount int, sampleRate int) *AudioBuffer
NewAudioBuffer creates a buffer from Go-managed memory
func ReadWaveMultiChannel ¶
func ReadWaveMultiChannel(filename string) *AudioBuffer
ReadWave reads from disk into C-managed memory (Zero-Copy) Note that you have to use AudioBuffer.Release() to avoid memory leak
func (*AudioBuffer) Release ¶
func (b *AudioBuffer) Release()
Release manually frees C-allocated memory
func (*AudioBuffer) Save ¶
func (b *AudioBuffer) Save(filename string) bool
type AudioEvent ¶
type AudioTagging ¶
type AudioTagging struct {
// contains filtered or unexported fields
}
func NewAudioTagging ¶
func NewAudioTagging(config *AudioTaggingConfig) *AudioTagging
The user is responsible to invoke DeleteAudioTagging() to free the returned tagger to avoid memory leak
func (*AudioTagging) Compute ¶
func (tagging *AudioTagging) Compute(s *OfflineStream, topK int32) []AudioEvent
type AudioTaggingConfig ¶
type AudioTaggingConfig struct {
Model AudioTaggingModelConfig
Labels string
TopK int32
}
type AudioTaggingModelConfig ¶
type AudioTaggingModelConfig struct {
Zipformer OfflineZipformerAudioTaggingModelConfig
Ced string
NumThreads int32
Debug int32
Provider string
}
type CircularBuffer ¶
type CircularBuffer struct {
// contains filtered or unexported fields
}
func NewCircularBuffer ¶
func NewCircularBuffer(capacity int) *CircularBuffer
func (*CircularBuffer) Head ¶
func (buffer *CircularBuffer) Head() int
func (*CircularBuffer) Pop ¶
func (buffer *CircularBuffer) Pop(n int)
func (*CircularBuffer) Push ¶
func (buffer *CircularBuffer) Push(samples []float32)
func (*CircularBuffer) Reset ¶
func (buffer *CircularBuffer) Reset()
func (*CircularBuffer) Size ¶
func (buffer *CircularBuffer) Size() int
type DenoisedAudio ¶
type DenoisedAudio struct {
// Normalized samples in the range [-1, 1]
Samples []float32
SampleRate int
}
func (*DenoisedAudio) Save ¶
func (audio *DenoisedAudio) Save(filename string) bool
type FastClusteringConfig ¶
type FeatureConfig ¶
type FeatureConfig struct {
// Sample rate expected by the model. It is 16000 for all
// pre-trained models provided by us
SampleRate int
// Feature dimension expected by the model. It is 80 for all
// pre-trained models provided by us
FeatureDim int
}
Configuration for the feature extractor
type GeneratedAudio ¶
type GeneratedAudio struct {
// Normalized samples in the range [-1, 1]
Samples []float32
SampleRate int
}
func (*GeneratedAudio) Save ¶
func (audio *GeneratedAudio) Save(filename string) bool
func (*GeneratedAudio) ToBuffer ¶
func (audio *GeneratedAudio) ToBuffer() []byte
type GenerationConfig ¶
type HomophoneReplacerConfig ¶
type KeywordSpotter ¶
type KeywordSpotter struct {
// contains filtered or unexported fields
}
func NewKeywordSpotter ¶
func NewKeywordSpotter(config *KeywordSpotterConfig) *KeywordSpotter
The user is responsible to invoke DeleteKeywordSpotter() to free the returned spotter to avoid memory leak
func (*KeywordSpotter) Decode ¶
func (spotter *KeywordSpotter) Decode(s *OnlineStream)
Decode the stream. Before calling this function, you have to ensure that spotter.IsReady(s) returns true. Otherwise, you will be SAD.
You usually use it like below:
for spotter.IsReady(s) {
spotter.Decode(s)
}
func (*KeywordSpotter) GetResult ¶
func (spotter *KeywordSpotter) GetResult(s *OnlineStream) *KeywordSpotterResult
Get the current result of stream since the last invoke of Reset()
func (*KeywordSpotter) IsReady ¶
func (spotter *KeywordSpotter) IsReady(s *OnlineStream) bool
Check whether the stream has enough feature frames for decoding. Return true if this stream is ready for decoding. Return false otherwise.
You will usually use it like below:
for spotter.IsReady(s) {
spotter.Decode(s)
}
func (*KeywordSpotter) Reset ¶
func (spotter *KeywordSpotter) Reset(s *OnlineStream)
You MUST call it right after detecting a keyword
type KeywordSpotterConfig ¶
type KeywordSpotterConfig struct {
FeatConfig FeatureConfig
ModelConfig OnlineModelConfig
MaxActivePaths int
KeywordsFile string
KeywordsScore float32
KeywordsThreshold float32
KeywordsBuf string
KeywordsBufSize int
}
Configuration for the online/streaming recognizer.
type KeywordSpotterResult ¶
type KeywordSpotterResult struct {
Keyword string
}
type OfflineDolphinModelConfig ¶
type OfflineDolphinModelConfig struct {
Model string // Path to the model, e.g., model.onnx or model.int8.onnx
}
type OfflineFireRedAsrCtcModelConfig ¶
type OfflineFireRedAsrCtcModelConfig struct {
Model string // Path to the model, e.g., model.onnx or model.int8.onnx
}
type OfflineLMConfig ¶
type OfflineLMConfig struct {
Model string // Path to the model
Scale float32 // scale for LM score
}
Configuration for offline LM.
type OfflineMedAsrCtcModelConfig ¶
type OfflineMedAsrCtcModelConfig struct {
Model string // Path to the model, e.g., model.onnx or model.int8.onnx
}
type OfflineModelConfig ¶
type OfflineModelConfig struct {
Transducer OfflineTransducerModelConfig
Paraformer OfflineParaformerModelConfig
NemoCTC OfflineNemoEncDecCtcModelConfig
Whisper OfflineWhisperModelConfig
Tdnn OfflineTdnnModelConfig
SenseVoice OfflineSenseVoiceModelConfig
Moonshine OfflineMoonshineModelConfig
FireRedAsr OfflineFireRedAsrModelConfig
FunAsrNano OfflineFunASRNanoModelConfig
Dolphin OfflineDolphinModelConfig
ZipformerCtc OfflineZipformerCtcModelConfig
Canary OfflineCanaryModelConfig
WenetCtc OfflineWenetCtcModelConfig
Omnilingual OfflineOmnilingualAsrCtcModelConfig
MedAsr OfflineMedAsrCtcModelConfig
FireRedAsrCtc OfflineFireRedAsrCtcModelConfig
Qwen3ASR OfflineQwen3ASRModelConfig
CohereTranscribe OfflineCohereTranscribeModelConfig
Tokens string // Path to tokens.txt
// Number of threads to use for neural network computation
NumThreads int
// 1 to print model meta information while loading
Debug int
// Optional. Valid values: cpu, cuda, coreml
Provider string
// Optional. Specify it for faster model initialization.
ModelType string
ModelingUnit string // Optional. cjkchar, bpe, cjkchar+bpe
BpeVocab string // Optional.
TeleSpeechCtc string // Optional.
}
type OfflineMoonshineModelConfig ¶
type OfflineMoonshineModelConfig struct {
Preprocessor string
Encoder string
UncachedDecoder string
CachedDecoder string
MergedDecoder string
}
For Moonshine v1, you need 4 models:
- preprocessor, encoder, uncached_decoder, cached_decoder
For Moonshine v2, you need 2 models:
- encoder, merged_decoder
type OfflineNemoEncDecCtcModelConfig ¶
type OfflineNemoEncDecCtcModelConfig struct {
Model string // Path to the model, e.g., model.onnx or model.int8.onnx
}
Configuration for offline/non-streaming NeMo CTC models.
Please refer to https://k2-fsa.github.io/sherpa/onnx/pretrained_models/offline-ctc/index.html to download pre-trained models
type OfflineOmnilingualAsrCtcModelConfig ¶
type OfflineOmnilingualAsrCtcModelConfig struct {
Model string // Path to the model, e.g., model.onnx or model.int8.onnx
}
type OfflineParaformerModelConfig ¶
type OfflineParaformerModelConfig struct {
Model string // Path to the model, e.g., model.onnx or model.int8.onnx
}
Configuration for offline/non-streaming paraformer.
please refer to https://k2-fsa.github.io/sherpa/onnx/pretrained_models/offline-paraformer/index.html to download pre-trained models
type OfflinePunctuation ¶
type OfflinePunctuation struct {
// contains filtered or unexported fields
}
func NewOfflinePunctuation ¶
func NewOfflinePunctuation(config *OfflinePunctuationConfig) *OfflinePunctuation
func (*OfflinePunctuation) AddPunct ¶
func (punc *OfflinePunctuation) AddPunct(text string) string
type OfflinePunctuationConfig ¶
type OfflinePunctuationConfig struct {
Model OfflinePunctuationModelConfig
}
type OfflinePunctuationModelConfig ¶
type OfflinePunctuationModelConfig struct {
CtTransformer string
NumThreads int
Debug int // true to print debug information of the model
Provider string
}
============================================================ For punctuation ============================================================
type OfflineRecognizer ¶
type OfflineRecognizer struct {
// contains filtered or unexported fields
}
It wraps a pointer from C
func NewOfflineRecognizer ¶
func NewOfflineRecognizer(config *OfflineRecognizerConfig) *OfflineRecognizer
The user is responsible to invoke DeleteOfflineRecognizer() to free the returned recognizer to avoid memory leak
func (*OfflineRecognizer) Decode ¶
func (recognizer *OfflineRecognizer) Decode(s *OfflineStream)
Decode the offline stream.
func (*OfflineRecognizer) DecodeStreams ¶
func (recognizer *OfflineRecognizer) DecodeStreams(s []*OfflineStream)
Decode multiple streams in parallel, i.e., in batch.
func (*OfflineRecognizer) SetConfig ¶
func (r *OfflineRecognizer) SetConfig(config *OfflineRecognizerConfig)
Set new config to replace
type OfflineRecognizerConfig ¶
type OfflineRecognizerConfig struct {
FeatConfig FeatureConfig
ModelConfig OfflineModelConfig
LmConfig OfflineLMConfig
// Valid decoding method: greedy_search, modified_beam_search
DecodingMethod string
// Used only when DecodingMethod is modified_beam_search.
MaxActivePaths int
HotwordsFile string
HotwordsScore float32
BlankPenalty float32
RuleFsts string
RuleFars string
Hr HomophoneReplacerConfig
}
Configuration for the offline/non-streaming recognizer.
type OfflineRecognizerResult ¶
type OfflineRecognizerResult struct {
Text string
Tokens []string
Timestamps []float32
Durations []float32
YsLogProbs []float32
Lang string
Emotion string
Event string
}
It contains recognition result of an offline stream.
type OfflineSourceSeparationConfig ¶
type OfflineSourceSeparationConfig struct {
Model OfflineSourceSeparationModelConfig
}
Config is the top-level configuration class
type OfflineSourceSeparationModelConfig ¶
type OfflineSourceSeparationModelConfig struct {
Spleeter OfflineSourceSeparationSpleeterModelConfig
Uvr OfflineSourceSeparationUvrModelConfig
NumThreads int
Debug bool
Provider string // e.g., "cpu", "cuda", "coreml"
}
type OfflineSourceSeparationUvrModelConfig ¶
type OfflineSourceSeparationUvrModelConfig struct {
Model string
}
UvrConfig wraps SherpaOnnxOfflineSourceSeparationUvrModelConfig
type OfflineSpeakerDiarization ¶
type OfflineSpeakerDiarization struct {
// contains filtered or unexported fields
}
func NewOfflineSpeakerDiarization ¶
func NewOfflineSpeakerDiarization(config *OfflineSpeakerDiarizationConfig) *OfflineSpeakerDiarization
func (*OfflineSpeakerDiarization) Process ¶
func (sd *OfflineSpeakerDiarization) Process(samples []float32) []OfflineSpeakerDiarizationSegment
func (*OfflineSpeakerDiarization) ProcessWithContext ¶
func (sd *OfflineSpeakerDiarization) ProcessWithContext( ctx context.Context, samples []float32, cb OfflineSpeakerDiarizationProgressCallback, ) ([]OfflineSpeakerDiarizationSegment, error)
ProcessWithContext runs offline speaker diarization, reports native progress, and returns promptly after the native implementation observes cancellation at a processing-chunk boundary.
func (*OfflineSpeakerDiarization) ProcessWithProgressCallback ¶
func (sd *OfflineSpeakerDiarization) ProcessWithProgressCallback( samples []float32, cb OfflineSpeakerDiarizationProgressCallback, ) []OfflineSpeakerDiarizationSegment
ProcessWithProgressCallback runs offline speaker diarization and reports native progress.
func (*OfflineSpeakerDiarization) SampleRate ¶
func (sd *OfflineSpeakerDiarization) SampleRate() int
func (*OfflineSpeakerDiarization) SetConfig ¶
func (sd *OfflineSpeakerDiarization) SetConfig(config *OfflineSpeakerDiarizationConfig)
only config.Clustering is used. All other fields are ignored
type OfflineSpeakerDiarizationConfig ¶
type OfflineSpeakerDiarizationConfig struct {
Segmentation OfflineSpeakerSegmentationModelConfig
Embedding SpeakerEmbeddingExtractorConfig
Clustering FastClusteringConfig
MinDurationOn float32
MinDurationOff float32
}
type OfflineSpeakerDiarizationProgressCallback ¶
OfflineSpeakerDiarizationProgressCallback receives processed and total normalized work units while offline speaker diarization is running.
type OfflineSpeakerSegmentationModelConfig ¶
type OfflineSpeakerSegmentationModelConfig struct {
Pyannote OfflineSpeakerSegmentationPyannoteModelConfig
NumThreads int
Debug int
Provider string
}
type OfflineSpeakerSegmentationPyannoteModelConfig ¶
type OfflineSpeakerSegmentationPyannoteModelConfig struct {
Model string
}
============================================================ For offline speaker diarization ============================================================
type OfflineSpeechDenoiser ¶
type OfflineSpeechDenoiser struct {
// contains filtered or unexported fields
}
func NewOfflineSpeechDenoiser ¶
func NewOfflineSpeechDenoiser(config *OfflineSpeechDenoiserConfig) *OfflineSpeechDenoiser
The user is responsible to invoke DeleteOfflineSpeechDenoiser() to free the returned tts to avoid memory leak
func (*OfflineSpeechDenoiser) Run ¶
func (sd *OfflineSpeechDenoiser) Run(samples []float32, sampleRate int) *DenoisedAudio
func (*OfflineSpeechDenoiser) SampleRate ¶
func (sd *OfflineSpeechDenoiser) SampleRate() int
type OfflineSpeechDenoiserConfig ¶
type OfflineSpeechDenoiserConfig struct {
Model OfflineSpeechDenoiserModelConfig
}
type OfflineSpeechDenoiserDpdfNetModelConfig ¶
type OfflineSpeechDenoiserDpdfNetModelConfig struct {
Model string
}
type OfflineSpeechDenoiserGtcrnModelConfig ¶
type OfflineSpeechDenoiserGtcrnModelConfig struct {
Model string
}
type OfflineSpeechDenoiserModelConfig ¶
type OfflineSpeechDenoiserModelConfig struct {
Gtcrn OfflineSpeechDenoiserGtcrnModelConfig
DpdfNet OfflineSpeechDenoiserDpdfNetModelConfig
NumThreads int32
Debug int32
Provider string
}
type OfflineStream ¶
type OfflineStream struct {
// contains filtered or unexported fields
}
It wraps a pointer from C
func NewAudioTaggingStream ¶
func NewAudioTaggingStream(tagging *AudioTagging) *OfflineStream
The user is responsible to invoke DeleteOfflineStream() to free the returned stream to avoid memory leak
func NewOfflineStream ¶
func NewOfflineStream(recognizer *OfflineRecognizer) *OfflineStream
The user is responsible to invoke DeleteOfflineStream() to free the returned stream to avoid memory leak
func (*OfflineStream) AcceptWaveform ¶
func (s *OfflineStream) AcceptWaveform(sampleRate int, samples []float32)
Input audio samples for the offline stream. Please only call it once. That is, input all samples at once.
sampleRate is the sample rate of the input audio samples. If it is different from the value expected by the feature extractor, we will do resampling inside.
samples contains the actual audio samples. Each sample is in the range [-1, 1].
func (*OfflineStream) GetOption ¶
func (s *OfflineStream) GetOption(key string) string
Get a key-value option from the offline stream. Returns an empty string if the option is not set.
func (*OfflineStream) GetResult ¶
func (s *OfflineStream) GetResult() *OfflineRecognizerResult
Get the recognition result of the offline stream.
func (*OfflineStream) HasOption ¶
func (s *OfflineStream) HasOption(key string) bool
Check whether the given option exists in the offline stream. Return true if the option exists. Return false otherwise.
func (*OfflineStream) SetOption ¶
func (s *OfflineStream) SetOption(key string, value string)
Set a key-value option on the offline stream. This provides a generic mechanism for passing per-stream runtime parameters to the recognizer (e.g., "task", "prompt").
type OfflineTdnnModelConfig ¶
type OfflineTdnnModelConfig struct {
Model string
}
type OfflineTransducerModelConfig ¶
type OfflineTransducerModelConfig struct {
Encoder string // Path to the encoder model, i.e., encoder.onnx or encoder.int8.onnx
Decoder string // Path to the decoder model
Joiner string // Path to the joiner model
}
Configuration for offline/non-streaming transducer.
Please refer to https://k2-fsa.github.io/sherpa/onnx/pretrained_models/offline-transducer/index.html to download pre-trained models
type OfflineTts ¶
type OfflineTts struct {
// contains filtered or unexported fields
}
The offline tts class. It wraps a pointer from C.
func NewOfflineTts ¶
func NewOfflineTts(config *OfflineTtsConfig) *OfflineTts
The user is responsible to invoke DeleteOfflineTts() to free the returned tts to avoid memory leak
func (*OfflineTts) Generate ¶
func (tts *OfflineTts) Generate(text string, sid int, speed float32) *GeneratedAudio
func (*OfflineTts) GenerateWithCallback ¶
func (tts *OfflineTts) GenerateWithCallback( text string, sid int, speed float32, cb sherpaOnnxGeneratedAudioCallbackWithArg, ) *GeneratedAudio
func (*OfflineTts) GenerateWithConfig ¶
func (tts *OfflineTts) GenerateWithConfig( text string, cfg *GenerationConfig, cb sherpaOnnxGeneratedAudioProgressCallbackWithArg, ) *GeneratedAudio
func (*OfflineTts) GenerateWithProgressCallback ¶
func (tts *OfflineTts) GenerateWithProgressCallback( text string, sid int, speed float32, cb sherpaOnnxGeneratedAudioProgressCallbackWithArg, ) *GeneratedAudio
func (*OfflineTts) GenerateWithZipvoice
deprecated
func (tts *OfflineTts) GenerateWithZipvoice( text, promptText string, promptSamples []float32, promptSampleRate int, speed float32, numSteps int, ) *GeneratedAudio
Deprecated: Use GenerateWithConfig() instead.
func (*OfflineTts) NumSpeakers ¶
func (tts *OfflineTts) NumSpeakers() int
func (*OfflineTts) SampleRate ¶
func (tts *OfflineTts) SampleRate() int
type OfflineTtsConfig ¶
type OfflineTtsConfig struct {
Model OfflineTtsModelConfig
RuleFsts string
RuleFars string
MaxNumSentences int
SilenceScale float32
}
type OfflineTtsKittenModelConfig ¶
type OfflineTtsKittenModelConfig struct {
Model string // Path to the model for kitten
Voices string // Path to the voices.bin for kitten
Tokens string // Path to tokens.txt
DataDir string // Path to espeak-ng-data directory
LengthScale float32 // Please use 1.0 in general. Smaller -> Faster speech speed. Larger -> Slower speech speed
}
type OfflineTtsKokoroModelConfig ¶
type OfflineTtsKokoroModelConfig struct {
Model string // Path to the model for kokoro
Voices string // Path to the voices.bin for kokoro
Tokens string // Path to tokens.txt
DataDir string // Path to espeak-ng-data directory
DictDir string // unused
Lexicon string // Path to lexicon files
Lang string // Example: es for Spanish, fr-fr for French. Can be empty
LengthScale float32 // Please use 1.0 in general. Smaller -> Faster speech speed. Larger -> Slower speech speed
}
type OfflineTtsMatchaModelConfig ¶
type OfflineTtsMatchaModelConfig struct {
AcousticModel string // Path to the acoustic model for MatchaTTS
Vocoder string // Path to the vocoder model for MatchaTTS
Lexicon string // Path to lexicon.txt
Tokens string // Path to tokens.txt
DataDir string // Path to espeak-ng-data directory
NoiseScale float32 // noise scale for vits models. Please use 0.667 in general
LengthScale float32 // Please use 1.0 in general. Smaller -> Faster speech speed. Larger -> Slower speech speed
DictDir string // unused
}
type OfflineTtsModelConfig ¶
type OfflineTtsModelConfig struct {
Vits OfflineTtsVitsModelConfig
Matcha OfflineTtsMatchaModelConfig
Kokoro OfflineTtsKokoroModelConfig
Kitten OfflineTtsKittenModelConfig
Zipvoice OfflineTtsZipvoiceModelConfig
Pocket OfflineTtsPocketModelConfig
Supertonic OfflineTtsSupertonicModelConfig
// Number of threads to use for neural network computation
NumThreads int
// 1 to print model meta information while loading
Debug int
// Optional. Valid values: cpu, cuda, coreml
Provider string
}
type OfflineTtsPocketModelConfig ¶
type OfflineTtsPocketModelConfig struct {
LmFlow string // lm_flow
LmMain string // lm_main
Encoder string // encoder
Decoder string // decoder
TextConditioner string // text_conditioner
VocabJson string // vocab_json
TokenScoresJson string // token_scores_json
VoiceEmbeddingCacheCapacity int // voice_embedding_cache_capacity
}
type OfflineTtsSupertonicModelConfig ¶
type OfflineTtsSupertonicModelConfig struct {
DurationPredictor string // Path to duration_predictor.onnx
TextEncoder string // Path to text_encoder.onnx
VectorEstimator string // Path to vector_estimator.onnx
Vocoder string // Path to vocoder.onnx
TtsJson string // Path to tts.json
UnicodeIndexer string // Path to unicode_indexer.bin
VoiceStyle string // Path to voice.bin
}
type OfflineTtsVitsModelConfig ¶
type OfflineTtsVitsModelConfig struct {
Model string // Path to the VITS onnx model
Lexicon string // Path to lexicon.txt
Tokens string // Path to tokens.txt
DataDir string // Path to espeak-ng-data directory
NoiseScale float32 // noise scale for vits models. Please use 0.667 in general
NoiseScaleW float32 // noise scale for vits models. Please use 0.8 in general
LengthScale float32 // Please use 1.0 in general. Smaller -> Faster speech speed. Larger -> Slower speech speed
DictDir string // unused
}
Configuration for offline/non-streaming text-to-speech (TTS).
Please refer to https://k2-fsa.github.io/sherpa/onnx/tts/pretrained_models/index.html to download pre-trained models
type OfflineTtsZipvoiceModelConfig ¶
type OfflineTtsZipvoiceModelConfig struct {
Tokens string // Path to tokens.txt for ZipVoice
Encoder string // Path to text encoder (e.g. encoder.onnx)
Decoder string // Path to flow-matching decoder (e.g. fm_decoder.onnx)
DataDir string // Path to espeak-ng-data
Lexicon string // Path to lexicon.txt (needed for zh)
Vocoder string // Path to vocoder (e.g. vocos_24khz.onnx)
FeatScale float32 // Feature scale
TShift float32 // t-shift (<1 shifts to smaller t)
TargetRms float32 // Target RMS for speech normalization
GuidanceScale float32 // CFG scale
}
type OfflineWenetCtcModelConfig ¶
type OfflineWenetCtcModelConfig struct {
Model string // Path to the model, e.g., model.onnx or model.int8.onnx
}
type OfflineZipformerAudioTaggingModelConfig ¶
type OfflineZipformerAudioTaggingModelConfig struct {
Model string
}
Configuration for the audio tagging.
type OfflineZipformerCtcModelConfig ¶
type OfflineZipformerCtcModelConfig struct {
Model string // Path to the model, e.g., model.onnx or model.int8.onnx
}
type OnlineModelConfig ¶
type OnlineModelConfig struct {
Transducer OnlineTransducerModelConfig
Paraformer OnlineParaformerModelConfig
Zipformer2Ctc OnlineZipformer2CtcModelConfig
NemoCtc OnlineNemoCtcModelConfig
ToneCtc OnlineToneCtcModelConfig
Tokens string // Path to tokens.txt
NumThreads int // Number of threads to use for neural network computation
Provider string // Optional. Valid values are: cpu, cuda, coreml
Debug int // 1 to show model meta information while loading it.
ModelType string // Optional. You can specify it for faster model initialization
ModelingUnit string // Optional. cjkchar, bpe, cjkchar+bpe
BpeVocab string // Optional.
TokensBuf string // Optional.
TokensBufSize int // Optional.
}
Configuration for online/streaming models
Please refer to https://k2-fsa.github.io/sherpa/onnx/pretrained_models/online-transducer/index.html https://k2-fsa.github.io/sherpa/onnx/pretrained_models/online-paraformer/index.html to download pre-trained models
type OnlineNemoCtcModelConfig ¶
type OnlineNemoCtcModelConfig struct {
Model string // Path to the onnx model
}
type OnlineParaformerModelConfig ¶
type OnlineParaformerModelConfig struct {
Encoder string // Path to the encoder model, e.g., encoder.onnx or encoder.int8.onnx
Decoder string // Path to the decoder model.
}
Configuration for online/streaming paraformer models
Please refer to https://k2-fsa.github.io/sherpa/onnx/pretrained_models/online-paraformer/index.html to download pre-trained models
type OnlinePunctuation ¶
type OnlinePunctuation struct {
// contains filtered or unexported fields
}
func NewOnlinePunctuation ¶
func NewOnlinePunctuation(config *OnlinePunctuationConfig) *OnlinePunctuation
func (*OnlinePunctuation) AddPunct ¶
func (punc *OnlinePunctuation) AddPunct(text string) string
type OnlinePunctuationConfig ¶
type OnlinePunctuationConfig struct {
Model OnlinePunctuationModelConfig
}
type OnlineRecognizer ¶
type OnlineRecognizer struct {
// contains filtered or unexported fields
}
The online recognizer class. It wraps a pointer from C.
func NewOnlineRecognizer ¶
func NewOnlineRecognizer(config *OnlineRecognizerConfig) *OnlineRecognizer
The user is responsible to invoke DeleteOnlineRecognizer() to free the returned recognizer to avoid memory leak
func (*OnlineRecognizer) Decode ¶
func (recognizer *OnlineRecognizer) Decode(s *OnlineStream)
Decode the stream. Before calling this function, you have to ensure that recognizer.IsReady(s) returns true. Otherwise, you will be SAD.
You usually use it like below:
for recognizer.IsReady(s) {
recognizer.Decode(s)
}
func (*OnlineRecognizer) DecodeStreams ¶
func (recognizer *OnlineRecognizer) DecodeStreams(s []*OnlineStream)
Decode multiple streams in parallel, i.e., in batch. You have to ensure that each stream is ready for decoding. Otherwise, you will be SAD.
func (*OnlineRecognizer) GetResult ¶
func (recognizer *OnlineRecognizer) GetResult(s *OnlineStream) *OnlineRecognizerResult
Get the current result of stream since the last invoke of Reset()
func (*OnlineRecognizer) IsEndpoint ¶
func (recognizer *OnlineRecognizer) IsEndpoint(s *OnlineStream) bool
Return true if an endpoint is detected.
You usually use it like below:
if recognizer.IsEndpoint(s) {
// do your own stuff after detecting an endpoint
recognizer.Reset(s)
}
func (*OnlineRecognizer) IsReady ¶
func (recognizer *OnlineRecognizer) IsReady(s *OnlineStream) bool
Check whether the stream has enough feature frames for decoding. Return true if this stream is ready for decoding. Return false otherwise.
You will usually use it like below:
for recognizer.IsReady(s) {
recognizer.Decode(s)
}
func (*OnlineRecognizer) Reset ¶
func (recognizer *OnlineRecognizer) Reset(s *OnlineStream)
After calling this function, the internal neural network model states are reset and IsEndpoint(s) would return false. GetResult(s) would also return an empty string.
type OnlineRecognizerConfig ¶
type OnlineRecognizerConfig struct {
FeatConfig FeatureConfig
ModelConfig OnlineModelConfig
// Valid decoding methods: greedy_search, modified_beam_search
DecodingMethod string
// Used only when DecodingMethod is modified_beam_search. It specifies
// the maximum number of paths to keep during the search
MaxActivePaths int
EnableEndpoint int // 1 to enable endpoint detection.
// Please see
// https://k2-fsa.github.io/sherpa/ncnn/endpoint.html
// for the meaning of Rule1MinTrailingSilence, Rule2MinTrailingSilence
// and Rule3MinUtteranceLength.
Rule1MinTrailingSilence float32
Rule2MinTrailingSilence float32
Rule3MinUtteranceLength float32
HotwordsFile string
HotwordsScore float32
BlankPenalty float32
CtcFstDecoderConfig OnlineCtcFstDecoderConfig
RuleFsts string
RuleFars string
HotwordsBuf string
HotwordsBufSize int
Hr HomophoneReplacerConfig
}
Configuration for the online/streaming recognizer.
type OnlineRecognizerResult ¶
It contains the recognition result for a online stream.
type OnlineSpeechDenoiser ¶
type OnlineSpeechDenoiser struct {
// contains filtered or unexported fields
}
func NewOnlineSpeechDenoiser ¶
func NewOnlineSpeechDenoiser(config *OnlineSpeechDenoiserConfig) *OnlineSpeechDenoiser
The user is responsible to invoke DeleteOnlineSpeechDenoiser() to free the returned denoiser to avoid memory leak.
func (*OnlineSpeechDenoiser) Flush ¶
func (sd *OnlineSpeechDenoiser) Flush() *DenoisedAudio
func (*OnlineSpeechDenoiser) FrameShiftInSamples ¶
func (sd *OnlineSpeechDenoiser) FrameShiftInSamples() int
func (*OnlineSpeechDenoiser) Reset ¶
func (sd *OnlineSpeechDenoiser) Reset()
func (*OnlineSpeechDenoiser) Run ¶
func (sd *OnlineSpeechDenoiser) Run(samples []float32, sampleRate int) *DenoisedAudio
func (*OnlineSpeechDenoiser) SampleRate ¶
func (sd *OnlineSpeechDenoiser) SampleRate() int
type OnlineSpeechDenoiserConfig ¶
type OnlineSpeechDenoiserConfig struct {
Model OfflineSpeechDenoiserModelConfig
}
type OnlineStream ¶
type OnlineStream struct {
// contains filtered or unexported fields
}
The online stream class. It wraps a pointer from C.
func NewKeywordStream ¶
func NewKeywordStream(spotter *KeywordSpotter) *OnlineStream
The user is responsible to invoke DeleteOnlineStream() to free the returned stream to avoid memory leak
func NewKeywordStreamWithKeywords ¶
func NewKeywordStreamWithKeywords(spotter *KeywordSpotter, keywords string) *OnlineStream
The user is responsible to invoke DeleteOnlineStream() to free the returned stream to avoid memory leak
func NewOnlineStream ¶
func NewOnlineStream(recognizer *OnlineRecognizer) *OnlineStream
The user is responsible to invoke DeleteOnlineStream() to free the returned stream to avoid memory leak
func (*OnlineStream) AcceptWaveform ¶
func (s *OnlineStream) AcceptWaveform(sampleRate int, samples []float32)
Input audio samples for the stream.
sampleRate is the actual sample rate of the input audio samples. If it is different from the sample rate expected by the feature extractor, we will do resampling inside.
samples contains audio samples. Each sample is in the range [-1, 1]
func (*OnlineStream) GetOption ¶
func (s *OnlineStream) GetOption(key string) string
Get a key-value option from the online stream. Returns an empty string if the option is not set.
func (*OnlineStream) HasOption ¶
func (s *OnlineStream) HasOption(key string) bool
Check whether the given option exists in the online stream. Return true if the option exists. Return false otherwise.
func (*OnlineStream) InputFinished ¶
func (s *OnlineStream) InputFinished()
Signal that there will be no incoming audio samples. After calling this function, you cannot call OnlineStream.AcceptWaveform any longer.
The main purpose of this function is to flush the remaining audio samples buffered inside for feature extraction.
func (*OnlineStream) SetOption ¶
func (s *OnlineStream) SetOption(key string, value string)
Set a key-value option on the online stream. This provides a generic mechanism for passing per-stream runtime parameters to the recognizer (e.g., "is_final" for streaming Paraformer).
type OnlineToneCtcModelConfig ¶
type OnlineToneCtcModelConfig struct {
Model string // Path to the onnx model
}
type OnlineTransducerModelConfig ¶
type OnlineTransducerModelConfig struct {
Encoder string // Path to the encoder model, e.g., encoder.onnx or encoder.int8.onnx
Decoder string // Path to the decoder model.
Joiner string // Path to the joiner model.
}
Configuration for online/streaming transducer models
Please refer to https://k2-fsa.github.io/sherpa/onnx/pretrained_models/online-transducer/index.html to download pre-trained models
type OnlineZipformer2CtcModelConfig ¶
type OnlineZipformer2CtcModelConfig struct {
Model string // Path to the onnx model
}
Please refer to https://k2-fsa.github.io/sherpa/onnx/pretrained_models/online-ctc/index.html to download pre-trained models
type SileroVadModelConfig ¶
type SileroVadModelConfig struct {
Model string
Threshold float32
MinSilenceDuration float32
MinSpeechDuration float32
WindowSize int
MaxSpeechDuration float32
}
============================================================ For VAD ============================================================
type SourceSeparator ¶
type SourceSeparator struct {
// contains filtered or unexported fields
}
func NewSourceSeparator ¶
func NewSourceSeparator(cfg OfflineSourceSeparationConfig) *SourceSeparator
Please use SourceSeparator.Delete() to avoid memory leak
func (*SourceSeparator) Delete ¶
func (ss *SourceSeparator) Delete()
func (*SourceSeparator) Process ¶
func (ss *SourceSeparator) Process(buf *AudioBuffer) []*AudioBuffer
type SpeakerEmbeddingExtractor ¶
type SpeakerEmbeddingExtractor struct {
// contains filtered or unexported fields
}
func NewSpeakerEmbeddingExtractor ¶
func NewSpeakerEmbeddingExtractor(config *SpeakerEmbeddingExtractorConfig) *SpeakerEmbeddingExtractor
The user has to invoke DeleteSpeakerEmbeddingExtractor() to free the returned value to avoid memory leak
func (*SpeakerEmbeddingExtractor) Compute ¶
func (ex *SpeakerEmbeddingExtractor) Compute(stream *OnlineStream) []float32
func (*SpeakerEmbeddingExtractor) CreateStream ¶
func (ex *SpeakerEmbeddingExtractor) CreateStream() *OnlineStream
The user is responsible to invoke DeleteOnlineStream() to free the returned stream to avoid memory leak
func (*SpeakerEmbeddingExtractor) Dim ¶
func (ex *SpeakerEmbeddingExtractor) Dim() int
func (*SpeakerEmbeddingExtractor) IsReady ¶
func (ex *SpeakerEmbeddingExtractor) IsReady(stream *OnlineStream) bool
type SpeakerEmbeddingManager ¶
type SpeakerEmbeddingManager struct {
// contains filtered or unexported fields
}
func NewSpeakerEmbeddingManager ¶
func NewSpeakerEmbeddingManager(dim int) *SpeakerEmbeddingManager
The user has to invoke DeleteSpeakerEmbeddingManager() to free the returned value to avoid memory leak
func (*SpeakerEmbeddingManager) AllSpeakers ¶
func (m *SpeakerEmbeddingManager) AllSpeakers() []string
func (*SpeakerEmbeddingManager) Contains ¶
func (m *SpeakerEmbeddingManager) Contains(name string) bool
func (*SpeakerEmbeddingManager) NumSpeakers ¶
func (m *SpeakerEmbeddingManager) NumSpeakers() int
func (*SpeakerEmbeddingManager) Register ¶
func (m *SpeakerEmbeddingManager) Register(name string, embedding []float32) bool
func (*SpeakerEmbeddingManager) RegisterV ¶
func (m *SpeakerEmbeddingManager) RegisterV(name string, embeddings [][]float32) bool
func (*SpeakerEmbeddingManager) Remove ¶
func (m *SpeakerEmbeddingManager) Remove(name string) bool
type SpeechSegment ¶
type SpokenLanguageIdentification ¶
type SpokenLanguageIdentification struct {
// contains filtered or unexported fields
}
func NewSpokenLanguageIdentification ¶
func NewSpokenLanguageIdentification(config *SpokenLanguageIdentificationConfig) *SpokenLanguageIdentification
func (*SpokenLanguageIdentification) Compute ¶
func (slid *SpokenLanguageIdentification) Compute(stream *OfflineStream) *SpokenLanguageIdentificationResult
func (*SpokenLanguageIdentification) CreateStream ¶
func (slid *SpokenLanguageIdentification) CreateStream() *OfflineStream
The user has to invoke DeleteOfflineStream() to free the returned value to avoid memory leak
type SpokenLanguageIdentificationConfig ¶
type SpokenLanguageIdentificationConfig struct {
Whisper SpokenLanguageIdentificationWhisperConfig
NumThreads int
Debug int
Provider string
}
type SpokenLanguageIdentificationResult ¶
type SpokenLanguageIdentificationResult struct {
Lang string
}
type TenVadModelConfig ¶
type VadModelConfig ¶
type VadModelConfig struct {
SileroVad SileroVadModelConfig
TenVad TenVadModelConfig
SampleRate int
NumThreads int
Provider string
Debug int
}
type VoiceActivityDetector ¶
type VoiceActivityDetector struct {
// contains filtered or unexported fields
}
func NewVoiceActivityDetector ¶
func NewVoiceActivityDetector(config *VadModelConfig, bufferSizeInSeconds float32) *VoiceActivityDetector
func (*VoiceActivityDetector) AcceptWaveform ¶
func (vad *VoiceActivityDetector) AcceptWaveform(samples []float32)
func (*VoiceActivityDetector) Clear ¶
func (vad *VoiceActivityDetector) Clear()
func (*VoiceActivityDetector) Flush ¶
func (vad *VoiceActivityDetector) Flush()
func (*VoiceActivityDetector) Front ¶
func (vad *VoiceActivityDetector) Front() *SpeechSegment
func (*VoiceActivityDetector) IsEmpty ¶
func (vad *VoiceActivityDetector) IsEmpty() bool
func (*VoiceActivityDetector) IsSpeech ¶
func (vad *VoiceActivityDetector) IsSpeech() bool
func (*VoiceActivityDetector) Pop ¶
func (vad *VoiceActivityDetector) Pop()
func (*VoiceActivityDetector) Reset ¶
func (vad *VoiceActivityDetector) Reset()