nemo_curator.models.asr.faster_whisper

View as Markdown

Faster-Whisper implementation of the shared ASR adapter.

Module Contents

Classes

NameDescription
FasterWhisperASRRun Faster-Whisper on mono 16 kHz waveforms prepared by ASRStage.

Functions

Data

_LANGUAGE_ALIASES

_TARGET_SAMPLE_RATE

API

class nemo_curator.models.asr.faster_whisper.FasterWhisperASR(
model_id: str = 'large-v3',
revision: str | None = None,
compute_type: str = 'float16',
beam_size: int = 5,
vad_filter: bool = True,
without_timestamps: bool = True,
cpu_compute_type: str = 'int8'
)
Dataclass

Run Faster-Whisper on mono 16 kHz waveforms prepared by ASRStage.

_model
Any | None = field(default=None, init=False, repr=False)
beam_size
int = 5
compute_type
str = 'float16'
cpu_compute_type
str = 'int8'
model_id
str = 'large-v3'
revision
str | None = None
vad_filter
bool = True
without_timestamps
bool = True
nemo_curator.models.asr.faster_whisper.FasterWhisperASR.__post_init__() -> None
nemo_curator.models.asr.faster_whisper.FasterWhisperASR.download_weights_on_node() -> None

Populate Faster-Whisper’s node-local cache without allocating a GPU.

nemo_curator.models.asr.faster_whisper.FasterWhisperASR.load_model(
num_gpus: int
) -> None
nemo_curator.models.asr.faster_whisper.FasterWhisperASR.transcribe_batch(
items: list[dict[str, typing.Any]]
nemo_curator.models.asr.faster_whisper.FasterWhisperASR.unload_model() -> None
nemo_curator.models.asr.faster_whisper._download_whisper_model(
model_id: str,
revision: str | None
) -> None
nemo_curator.models.asr.faster_whisper._faster_whisper_stack() -> tuple[type, typing.Any]
nemo_curator.models.asr.faster_whisper._whisper_model_class() -> type
nemo_curator.models.asr.faster_whisper._LANGUAGE_ALIASES = {'fil': 'tl', 'in': 'id', 'iw': 'he', 'ji': 'yi', 'jv': 'jw', 'nb': 'no', 'tl': ...
nemo_curator.models.asr.faster_whisper._TARGET_SAMPLE_RATE = 16000