nemo_curator.models.asr.indic_canary

View as Markdown

Adapter for an Indic Canary model exported as a TensorRT-LLM engine.

Module Contents

Classes

NameDescription
IndicCanaryTRTLLMASRRun static-batch Indic Canary inference from a prebuilt engine directory.

Data

_DEFAULT_MAX_DURATION_SEC

_DEFAULT_MIN_DURATION_SEC

_MIN_DURATION_SAMPLES

_REQUIRED_ENGINE_FILES

_TARGET_SAMPLE_RATE

API

class nemo_curator.models.asr.indic_canary.IndicCanaryTRTLLMASR(
engine_dir: str,
num_beams: int = 4,
max_new_tokens: int = 374,
pnc: bool = False,
max_duration_sec: float = _DEFAULT_MAX_DURATION_SEC,
min_duration_sec: float = _DEFAULT_MIN_DURATION_SEC,
kv_cache_free_gpu_memory_fraction: float = 0.2,
cross_kv_cache_fraction: float = 0.2
)

Run static-batch Indic Canary inference from a prebuilt engine directory.

cross_kv_cache_fraction
= float(cross_kv_cache_fraction)
kv_cache_free_gpu_memory_fraction
= float(kv_cache_free_gpu_memory_fraction)
max_new_tokens
= int(max_new_tokens)
max_samples
= int(self.max_duration_sec * _TARGET_SAMPLE_RATE)
min_duration_sec
= min(min_duration_sec, self.max_duration_sec)
min_samples
= int(self.min_duration_sec * _TARGET_SAMPLE_RATE)
num_beams
= int(num_beams)
pnc
= bool(pnc)
nemo_curator.models.asr.indic_canary.IndicCanaryTRTLLMASR._normalize_language(
language: str
) -> str | None
nemo_curator.models.asr.indic_canary.IndicCanaryTRTLLMASR._prompt_config(
language: str
) -> dict[str, object]
nemo_curator.models.asr.indic_canary.IndicCanaryTRTLLMASR._transcribe_prepared(
padded: list[typing.Any],
durations: list[int],
prompts: list[dict[str, object]]
) -> list[str]

Run engine-sized sub-batches without exposing a second batch control.

nemo_curator.models.asr.indic_canary.IndicCanaryTRTLLMASR.download_weights_on_node() -> None

Validate local engine artifacts without allocating GPU state.

nemo_curator.models.asr.indic_canary.IndicCanaryTRTLLMASR.load_model(
num_gpus: int
) -> None

Load the TensorRT-LLM runtime on its one required GPU.

nemo_curator.models.asr.indic_canary.IndicCanaryTRTLLMASR.transcribe_batch(
items: list[dict[str, typing.Any]]

Transcribe supported rows and preserve their original positions.

nemo_curator.models.asr.indic_canary.IndicCanaryTRTLLMASR.unload_model() -> None

Release the engine and its CUDA allocations.

nemo_curator.models.asr.indic_canary._DEFAULT_MAX_DURATION_SEC = 40.0
nemo_curator.models.asr.indic_canary._DEFAULT_MIN_DURATION_SEC = 0.5
nemo_curator.models.asr.indic_canary._MIN_DURATION_SAMPLES = 400
nemo_curator.models.asr.indic_canary._REQUIRED_ENGINE_FILES = ('encoder/encoder.plan', 'encoder/config.json', 'decoder/config.json', 'decoder/...
nemo_curator.models.asr.indic_canary._TARGET_SAMPLE_RATE = 16000