class nemo_automodel.components.models.kimi_k3.config.KimiK3TextConfig(
vocab_size: int = 163840,
hidden_size: int = 7168,
head_dim: int | None = None,
intermediate_size: int = 33792,
num_hidden_layers: int = 93,
num_attention_heads: int = 96,
num_key_value_heads: int | None = None,
hidden_act: str = 'situ',
initializer_range: float = 0.02,
rms_norm_eps: float = 1e-05,
use_cache: bool = True,
pad_token_id: int = 0,
bos_token_id: int = 1,
eos_token_id: int = 2,
architectures: list[str] | None = None,
rope_theta: float = 10000.0,
rope_scaling: dict[str, typing.Any] | None = None,
tie_word_embeddings: bool = False,
attention_dropout: float = 0.0,
max_position_embeddings: int = 1048576,
moe_intermediate_size: int | None = 3072,
moe_renormalize: bool = True,
moe_router_activation_func: str = 'sigmoid',
num_experts: int | None = 896,
num_experts_per_token: int | None = 16,
num_shared_experts: int = 2,
routed_scaling_factor: float = 1.0,
first_k_dense_replace: int = 1,
moe_layer_freq: int = 1,
use_grouped_topk: bool = True,
num_expert_group: int = 1,
topk_group: int = 1,
topk_method: str = 'noaux_tc',
routed_expert_hidden_size: int | None = 3584,
latent_moe_use_norm: bool = True,
q_lora_rank: int | None = 1536,
kv_lora_rank: int | None = 512,
qk_nope_head_dim: int | None = 128,
qk_rope_head_dim: int | None = 64,
v_head_dim: int | None = 128,
mla_use_nope: bool = True,
mla_use_output_gate: bool = True,
linear_attn_config: dict[str, typing.Any] | None = None,
kda_mode: str = 'chunk',
kda_unpad_inputs: bool = True,
kda_use_fused_gate: bool = True,
kda_chunk_impl: str = 'fla',
kda_use_qk_l2norm_in_kernel: bool = True,
kda_disable_recompute: bool = False,
kda_conv_backend: str = 'triton',
kda_transpose_state_layout: bool = True,
situ_backend: str = 'torch',
attn_res_triton: bool = False,
attn_res_block_size: int | None = 12,
activation_situ_beta: float | None = 4.0,
activation_situ_linear_beta: float | None = 25.0,
num_nextn_predict_layers: int = 0,
kwargs: typing.Any = {}
)