nemo_gym.orchestration.api

View as Markdown

Module Contents

Classes

NameDescription
BaseComputeConfig-
BaseModelServiceConfigBase for services that serve a model and can be wired as the policy model.
BaseServiceConfig-
BenchmarkRunConfig-
DriverConfig-
GymInstallConfig-
HealthCheckConfig-
JobConfig-
NodePool-
OtelConfigAn OpenTelemetry collector beside every benchmark job: scrapes each model service’s
RayServiceConfig-
SlurmComputeConfig-
SubmitConfig-
VllmServiceConfig-
_StrictModel-

Functions

NameDescription
effective_ray_serveWhether the Ray Serve gateway manages this service’s instances/routing instead of vLLM’s own DP.
resolve_env_dictResolve lit:/host:/runtime: prefixes on env values. Every value must use one

Data

ComputeConfig

RUNTIME_ENV_PREFIX

ServiceConfig

_ENV_VAR_NAME_RE

API

class nemo_gym.orchestration.api.BaseComputeConfig()
class nemo_gym.orchestration.api.BaseModelServiceConfig()

Bases: BaseServiceConfig

Base for services that serve a model and can be wired as the policy model.

model
str
port
int = 8000
served_model_name
str | None = None
class nemo_gym.orchestration.api.BaseServiceConfig()

Bases: _StrictModel

container
str
env
dict[str, str] = {}
health_check
HealthCheckConfig | None = None
mounts
list[str] = []
node_pool
str | None = None
placement
str | None = None
pre_command
str = ''
nemo_gym.orchestration.api.BaseServiceConfig._resolve_env_prefixes(
v: dict[str, str]
) -> dict[str, str]
classmethod
class nemo_gym.orchestration.api.BenchmarkRunConfig()

Bases: _StrictModel

prepare
dict[str, Any] = {}
run
dict[str, Any] = {}
class nemo_gym.orchestration.api.DriverConfig()

Bases: _StrictModel

benchmarks
dict[str, BenchmarkRunConfig]
container
str = 'python:3.12'
env
dict[str, str] = {}
gym_install
GymInstallConfig | None = None
mounts
list[str] = []
policy_model
str | None = None
policy_model_type
str = 'openai_model'
nemo_gym.orchestration.api.DriverConfig._resolve_env_prefixes(
v: dict[str, str]
) -> dict[str, str]
classmethod
class nemo_gym.orchestration.api.GymInstallConfig()

Bases: _StrictModel

ref
str
repo
str = 'https://github.com/NVIDIA-NeMo/gym'
class nemo_gym.orchestration.api.HealthCheckConfig()

Bases: _StrictModel

path
str = '/health'
port
int | None = None
timeout_seconds
int = 60
class nemo_gym.orchestration.api.JobConfig()

Bases: _StrictModel

output_path
str
class nemo_gym.orchestration.api.NodePool()

Bases: _StrictModel

extra_args
dict[str, str] = {}
gpus_per_node
int | None = None
nodes
int = 1
ntasks_per_node
int = 1
partition
str
class nemo_gym.orchestration.api.OtelConfig()

Bases: _StrictModel

An OpenTelemetry collector beside every benchmark job: scrapes each model service’s Prometheus /metrics, receives OTLP from the job’s own processes on :4317/:4318, and ships both to an OTLP/HTTP backend while keeping a copy under <job dir>/otel/. On by default, so a run is observable unless it opts out; endpoint and service_name come from the deployment’s own config (a cluster fragment, typically) and are required while enabled.

binary
str = 'otelcol-contrib'
component
str = 'gym-vllm'
container
str | None = None
enabled
bool = True
endpoint
str | None = None
gpu_metrics_port
int | None = 9400
health_check_timeout_seconds
int = 300
node_metrics_port
int | None = 9100
scrape_interval_seconds
int = 15
service_name
str | None = None
token_env
str = 'OTEL_TOKEN'
nemo_gym.orchestration.api.OtelConfig._validate_positive(
v: int
) -> int
classmethod
nemo_gym.orchestration.api.OtelConfig._validate_token_env(
v: str
) -> str
classmethod
class nemo_gym.orchestration.api.RayServiceConfig()

Bases: BaseServiceConfig

extra_args
str = ''
node_pools
list[str] = []
num_cpus
int | None = None
num_gpus
int | None = None
port
int = 6379
resources
dict[str, dict[str, float]] = {}
type
Literal['ray']
class nemo_gym.orchestration.api.SlurmComputeConfig()

Bases: BaseComputeConfig

account
str
extra_args
dict[str, str] = {}
hostname
str | None = None
node_pools
dict[str, NodePool] = {}
type
Literal['slurm']
walltime
str | None = None
class nemo_gym.orchestration.api.SubmitConfig()

Bases: _StrictModel

compute
dict[str, ComputeConfig]
driver
DriverConfig
job
JobConfig
otel
OtelConfig = OtelConfig()
services
dict[str, ServiceConfig]
nemo_gym.orchestration.api.SubmitConfig._validate_vllm_gpu_footprint(
service_name: str,
total_nodes: int,
node_pools: dict[str, nemo_gym.orchestration.api.NodePool],
gpus_per_node_values: list[int],
is_ray_serve: bool
) -> None
class nemo_gym.orchestration.api.VllmServiceConfig()

Bases: BaseModelServiceConfig

extra_args
str = ''
number_of_instances
int = 1
pipeline_parallel_size
int = 1
tensor_parallel_size
int = 1
trust_remote_code
bool = False
type
Literal['vllm']
use_ray_serve
bool = False
nemo_gym.orchestration.api.VllmServiceConfig._validate_number_of_instances(
v: int
) -> int
classmethod
class nemo_gym.orchestration.api._StrictModel()

Bases: BaseModel

model_config
= ConfigDict(extra='forbid')
nemo_gym.orchestration.api.effective_ray_serve(
total_nodes: int,
gpus_per_node_values: list[int]
) -> bool

Whether the Ray Serve gateway manages this service’s instances/routing instead of vLLM’s own DP.

nemo_gym.orchestration.api.resolve_env_dict(
env: dict[str, str]
) -> dict[str, str]

Resolve lit:/host:/runtime: prefixes on env values. Every value must use one of these prefixes; a missing or misspelled prefix raises rather than being guessed at.

  • lit:VALUE -> literal VALUE.
  • host:VAR -> read from os.environ[VAR] on the machine running gym eval submit; raises if VAR isn’t set there.
  • runtime:VAR -> left unresolved; canonicalized to runtime:VAR for executors to pick up and reference from the job’s own environment at run time.
nemo_gym.orchestration.api.ComputeConfig = Annotated[Annotated[SlurmComputeConfig, Tag('slurm')], Discriminator('type')]
nemo_gym.orchestration.api.RUNTIME_ENV_PREFIX = 'runtime:'
nemo_gym.orchestration.api.ServiceConfig = Annotated[Annotated[VllmServiceConfig, Tag('vllm')] | Annotated[RayServiceConfig...
nemo_gym.orchestration.api._ENV_VAR_NAME_RE = re.compile('^[A-Za-z_][A-Za-z0-9_]*$')