nemo_automodel.components.models.laguna

View as Markdown

Submodules

Package Contents

Classes

NameDescription
LagunaConfigConfiguration for Poolside Laguna causal language models.
LagunaForCausalLMCausal LM wrapper for Laguna with Automodel checkpoint adapters.
LagunaModelBackbone model for Laguna SFT.

Data

ModelClass

API

class nemo_automodel.components.models.laguna.config.LagunaConfig(
vocab_size: int = 100352,
hidden_size: int = 2048,
intermediate_size: int = 8192,
num_hidden_layers: int = 48,
num_attention_heads: int = 32,
num_key_value_heads: int = 8,
head_dim: int = 128,
qkv_bias: bool = False,
attention_bias: bool = False,
gating: bool | str = True,
gating_types: list[str] | None = None,
hidden_act: str = 'silu',
max_position_embeddings: int = 4096,
initializer_range: float = 0.02,
rms_norm_eps: float = 1e-06,
use_cache: bool = True,
tie_word_embeddings: bool = False,
rope_parameters: dict[str, typing.Any] | None = None,
partial_rotary_factor: float | None = None,
attention_dropout: float = 0.0,
sliding_window: int | None = None,
layer_types: list[str] | None = None,
num_attention_heads_per_layer: list[int] | None = None,
swa_attention_sink_enabled: bool = False,
swa_rope_parameters: dict[str, typing.Any] | None = None,
num_experts: int = 256,
num_experts_per_tok: int = 16,
moe_intermediate_size: int = 1024,
shared_expert_intermediate_size: int = 1024,
norm_topk_prob: bool = True,
decoder_sparse_step: int = 1,
mlp_only_layers: list[int] | None = None,
mlp_layer_types: list[str] | None = None,
router_aux_loss_coef: float = 0.001,
moe_routed_scaling_factor: float = 1.0,
moe_apply_router_weight_on_input: bool = False,
moe_router_logit_softcapping: float = 0.0,
output_router_logits: bool = False,
torch_dtype: str = 'bfloat16',
kwargs = {}
)

Bases: PretrainedConfig

Configuration for Poolside Laguna causal language models.

bos_token_id
int | None = None
eos_token_id
int | list[int] | None = None
keys_to_ignore_at_inference
= ['past_key_values']
model_type
= 'laguna'
pad_token_id
int | None = None
nemo_automodel.components.models.laguna.config.LagunaConfig.from_dict(
config_dict: dict[str, typing.Any],
kwargs = {}
)
classmethod
class nemo_automodel.components.models.laguna.model.LagunaForCausalLM(
kwargs = {}
)

Bases: HFCheckpointingMixin, Module, MoEFSDPSyncMixin

Causal LM wrapper for Laguna with Automodel checkpoint adapters.

_keep_in_fp32_modules_strict
= ['mlp.gate.e_score_correction_bias', 'rotary_emb']
_uses_hf_attention
bool = True
backend
= backend or BackendConfig()
lm_head
model
state_dict_adapter
tie_word_embeddings_support
TieSupport = TieSupport.UNTIED_ONLY
nemo_automodel.components.models.laguna.model.LagunaForCausalLM.forward(
input_ids: torch.Tensor | None = None,
inputs_embeds: torch.Tensor | None = None,
position_ids: torch.Tensor | None = None,
attention_mask: torch.Tensor | dict[str, torch.Tensor] | None = None,
padding_mask: torch.Tensor | None = None,
past_key_values: typing.Any = None,
use_cache: bool | None = None,
logits_to_keep: typing.Union[int, torch.Tensor] = 0,
output_hidden_states: bool | None = None,
kwargs: typing.Any = {}
) -> transformers.modeling_outputs.CausalLMOutputWithPast

Run the Laguna causal language model.

Parameters:

input_ids
torch.Tensor | NoneDefaults to None

Optional token IDs of shape [batch, sequence].

inputs_embeds
torch.Tensor | NoneDefaults to None

Optional embeddings of shape [batch, sequence, hidden].

position_ids
torch.Tensor | NoneDefaults to None

Optional position IDs of shape [batch, sequence].

attention_mask
torch.Tensor | dict[str, torch.Tensor] | NoneDefaults to None

Optional 2D bool/int mask of shape [batch, sequence], a 4D additive mask, or a mapping with per-attention-type masks keyed by “full_attention” and “sliding_attention”.

padding_mask
torch.Tensor | NoneDefaults to None

Optional bool tensor of shape [batch, sequence], where True marks tokens excluded from MoE routing.

past_key_values
AnyDefaults to None

Unsupported KV-cache state.

use_cache
bool | NoneDefaults to None

Unsupported cache flag.

logits_to_keep
Union[int, torch.Tensor]Defaults to 0

If 0, compute logits for all sequence positions. If an int or tensor, compute logits only for the selected trailing positions.

output_hidden_states
bool | NoneDefaults to None

When true, include final hidden states in the output.

**kwargs
AnyDefaults to {}

Additional attention backend arguments.

Returns: CausalLMOutputWithPast

Causal LM output with logits and optional hidden states.

nemo_automodel.components.models.laguna.model.LagunaForCausalLM.from_config(
kwargs = {}
)
classmethod
nemo_automodel.components.models.laguna.model.LagunaForCausalLM.from_pretrained(
pretrained_model_name_or_path: str,
model_args = (),
kwargs = {}
)
classmethod
nemo_automodel.components.models.laguna.model.LagunaForCausalLM.get_input_embeddings()
nemo_automodel.components.models.laguna.model.LagunaForCausalLM.get_output_embeddings()
nemo_automodel.components.models.laguna.model.LagunaForCausalLM.initialize_weights(
buffer_device: torch.device | None = None,
dtype: torch.dtype = torch.bfloat16
) -> None
nemo_automodel.components.models.laguna.model.LagunaForCausalLM.set_input_embeddings(
value
)
nemo_automodel.components.models.laguna.model.LagunaForCausalLM.set_output_embeddings(
new_embeddings
)
nemo_automodel.components.models.laguna.model.LagunaForCausalLM.update_moe_gate_bias() -> None
class nemo_automodel.components.models.laguna.model.LagunaModel(
moe_overrides: dict | None = None
)

Bases: Module

Backbone model for Laguna SFT.

embed_tokens
has_sliding_layers
= 'sliding_attention' in config.layer_types
layers
moe_config
= moe_config or MoEConfig(**moe_defaults)
norm
rotary_emb
swa_rotary_emb
nemo_automodel.components.models.laguna.model.LagunaModel._build_causal_mask_mapping(
inputs_embeds: torch.Tensor,
attention_mask: torch.Tensor | dict[str, torch.Tensor] | None,
position_ids: torch.Tensor
) -> dict[str, torch.Tensor]

Build additive attention masks for full and sliding Laguna layers.

Parameters:

inputs_embeds
torch.Tensor

Input embedding tensor of shape [batch, sequence, hidden].

attention_mask
torch.Tensor | dict[str, torch.Tensor] | None

Optional sequence mask of shape [batch, sequence], 4D bool/additive mask, or mapping with masks keyed by “full_attention” and “sliding_attention”.

position_ids
torch.Tensor

Position tensor of shape [batch, sequence].

Returns: dict[str, torch.Tensor]

Mapping from attention type to additive masks of shape

nemo_automodel.components.models.laguna.model.LagunaModel.forward(
input_ids: torch.Tensor | None = None,
inputs_embeds: torch.Tensor | None = None,
position_ids: torch.Tensor | None = None,
attention_mask: torch.Tensor | dict[str, torch.Tensor] | None = None,
padding_mask: torch.Tensor | None = None,
kwargs: typing.Any = {}
) -> torch.Tensor

Run the Laguna decoder stack.

Parameters:

input_ids
torch.Tensor | NoneDefaults to None

Optional token IDs of shape [batch, sequence].

inputs_embeds
torch.Tensor | NoneDefaults to None

Optional embeddings of shape [batch, sequence, hidden].

position_ids
torch.Tensor | NoneDefaults to None

Optional position IDs of shape [batch, sequence].

attention_mask
torch.Tensor | dict[str, torch.Tensor] | NoneDefaults to None

Optional 2D bool/int mask of shape [batch, sequence], a 4D additive mask, or a mapping with per-attention-type masks keyed by “full_attention” and “sliding_attention”.

padding_mask
torch.Tensor | NoneDefaults to None

Optional bool tensor of shape [batch, sequence], where True marks tokens excluded from MoE routing.

**kwargs
AnyDefaults to {}

Additional attention backend arguments.

Returns: torch.Tensor

Final hidden states of shape [batch, sequence, hidden].

nemo_automodel.components.models.laguna.model.LagunaModel.init_weights(
buffer_device: torch.device | None = None
) -> None
nemo_automodel.components.models.laguna.model.ModelClass = LagunaForCausalLM