Skip to main content

Authoring a Custom Environment

The MochiEnv class is the base class for all SuperDex Gym environments. It provides standardized interfaces for physics simulation, observation and action spaces, and reward computation.

Smallest Complete Workflow​

Create one *_env.py module, define its typed configuration, load a scene, implement the three required hooks, then register and instantiate it. This example uses the shipped CartPole prefab so it can be reset and stepped without placeholder actor parameters.

1. Define the configuration class​

Define a derived config class so the extra fields are typed and documented:

from superdex.physics.utils.configclasses import configclass
from superdex.lab.gym.envs import MochiEnvCfg


@configclass
class MyCustomEnvCfg(MochiEnvCfg):
"""Options for MyCustomEnv."""

# Base options with environment-appropriate defaults.
control_frequency: int = 20
simulation_frequency: int = 100
steps_per_episode: int = 1000
reset_noise_scale: float = 0.1

# Custom environment options.
my_reward_weight: float = 1.0
"""Weight applied to the task reward term."""

use_gravity: bool = True
"""If False, gravity is disabled to simplify the environment."""

2. Implement the environment class​

The class loads the prefab, captures the initial agent state required by reset(), and implements the environment interface:

from typing import Any

import numpy as np
import superdex.physics as physics
from superdex.lab.gym.envs import (
ActionSpace,
Info,
MochiEnv,
ObservationSpace,
RewardTerms,
StructuredAction,
StructuredObservation,
)
from superdex.lab.gym.utils import mochi_helpers
from superdex.physics.paths import get_assets_root
from superdex.physics.utils.decorators import override_from


class MyCustomEnv(MochiEnv):
"""A minimal custom environment."""

def __init__(self, cfg: MyCustomEnvCfg | dict[str, Any]):
if not isinstance(cfg, MyCustomEnvCfg):
cfg = MyCustomEnvCfg(**cfg)

# Validate the configuration and initialize the physics context and,
# when configured, the renderer.
super().__init__(cfg)

self._my_reward_weight = cfg.my_reward_weight

# Create the scene and record the initial agent state.
self._init_scene(cfg)

# Derive the observation size from the agent that was just loaded.
num_dofs = self._agent.get_num_dofs()

self._setup_action_space(
control=ActionSpace(-1.0, 1.0, (1,), dtype=np.float32),
)
self._setup_observation_space(
agent_pose=ObservationSpace(-np.inf, np.inf, (num_dofs,), dtype=np.float32),
agent_velocity=ObservationSpace(
-np.inf, np.inf, (num_dofs,), dtype=np.float32
),
)

def _init_scene(self, cfg: MyCustomEnvCfg):
def scene_builder():
prefab_path = (
get_assets_root() / "benchmarks" / "cart_pole" / "cart_pole.mochi_scene"
)
prefab = physics.prefab.shallow_load_from_file(str(prefab_path))

if not cfg.use_gravity:
prefab.scene.gravity = [0, 0, 0]

params = mochi_helpers.PrefabParams()
params.agent_actor_name = "CartPole"
params.add_ground_plane = False
return mochi_helpers.init_prefab_scene(prefab, params)

# Include every field that changes scene construction in the shared-scene key.
self._load_scene(f"scene_{hash((cfg.use_gravity,))}", scene_builder)

# Required by the default _reset_scene().
self._initial_pose = mochi_helpers.get_articulated_pose(self._agent)
self._initial_velocity = mochi_helpers.get_articulated_joint_velocities(
self._agent
)

@override_from(MochiEnv)
def _apply_action(self, action: StructuredAction):
"""Apply the agent action to the simulation. Runs once per simulation substep."""
control = action["control"]

# Force control: apply a generalized force to the CartPole cart DOF.
force_scale = 100.0
self._agent.set_external_forces_on_dofs([0], [control.item() * force_scale])

@override_from(MochiEnv)
def _make_observation(self) -> tuple[StructuredObservation, Info]:
"""Sample the environment state and build the observation."""
agent_pose = mochi_helpers.get_articulated_pose(self._agent)
agent_velocity = mochi_helpers.get_articulated_joint_velocities(self._agent)

obs = {
"agent_pose": agent_pose,
"agent_velocity": agent_velocity,
}

info = {
"is_healthy": bool(abs(agent_pose[1]) <= 0.2),
}

return obs, info

@override_from(MochiEnv)
def _compute_reward_terms(
self,
action: StructuredAction,
observation: StructuredObservation,
info: Info,
) -> RewardTerms:
"""Compute the named reward components."""
control = action["control"]
control_cost = -0.1 * float(np.dot(control, control))
task_reward = 1.0 if info["is_healthy"] else 0.0

return {
"task": task_reward * self._my_reward_weight,
"ctrl": control_cost,
}

@override_from(MochiEnv)
def _check_stop_criteria(
self,
observation: StructuredObservation,
action: StructuredAction,
reward: RewardTerms,
info: Info,
):
"""Check for episode termination."""
# The base implementation truncates at steps_per_episode. Call it by keyword.
super()._check_stop_criteria(
action=action,
observation=observation,
reward=reward,
info=info,
)

if not info["is_healthy"]:
info["terminated_reason"] = "Pole left the upright range"
self._terminated = True

3. Register and run the environment​

Environments are registered explicitly -- there is no filesystem scanning. Declare an EnvSpec and register it before calling gym.make. Every spec shares the same entry_point -- the make_superdex_env factory -- and names the concrete class in kwargs["env_cls"] ("<module>:<ClassName>"); the canonical id lives in the superdex_gym namespace and ends in -vN, and the per-env configuration is kwargs["cfg"]:

import gymnasium as gym
from gymnasium.envs.registration import EnvSpec
from gymnasium.utils.env_checker import check_env

from superdex.lab.gym.registration import register_env_spec

register_env_spec(
EnvSpec(
id="superdex_gym/MyCustom-v0",
entry_point="superdex.lab.gym.registration:make_superdex_env",
kwargs={
"env_cls": "my_package.my_custom_env:MyCustomEnv",
"cfg": {},
},
)
)
env = gym.make("superdex_gym/MyCustom-v0")
try:
# Raises if the environment violates the Gymnasium API.
check_env(env.unwrapped, skip_render_check=True)

observation, info = env.reset(seed=0)
action = env.action_space.sample()
next_observation, reward, terminated, truncated, info = env.step(action)
finally:
env.close()

The checker validates the core Gymnasium contract.

Expected result​

reset() returns a flat NumPy observation and an info dict. step() accepts a flat action sampled from env.action_space and returns the next flat observation, the scalar sum of the named reward terms, terminated, truncated, and an updated info dict containing reward_task and reward_ctrl. With this example's steps_per_episode=1000, the base class truncates the episode at the time limit. The episode terminates earlier if the pole angle leaves the upright range [-0.2, 0.2] radians.

Authoring Contract​

For the shared reset, step, render, and close behavior; flattened spaces; conversion helpers; and runtime introspection, see the Environment Class API.

Within step(), operations occur in this order:

  1. Clip the action to the action space and convert it to its structured form.
  2. Apply the action and advance the scene for each simulation substep.
  3. Build the new observation and info.
  4. Compute reward terms from the new observation.
  5. Check the stop criteria.
  6. Save the step result and, in human render mode, update the renderer.

Required hooks​

Every concrete subclass implements exactly three abstract methods:

MethodPurpose
_apply_action(action)Apply the structured action to the simulation.
_make_observation()Return a structured observation and an info dict.
_compute_reward_terms(action, observation, info)Return named reward components. Their sum is the scalar reward, and each component is copied to info as reward_<name>.

Call _setup_action_space and _setup_observation_space once in __init__. Inside the hooks, actions and observations are structured dictionaries; at the Gymnasium boundary, MochiEnv converts them to flat arrays.

Capture the initial agent state

The default _reset_scene() requires _initial_pose and _initial_velocity. Set both immediately after loading the scene, as in the example above. The first reset() raises RuntimeError if either is missing.

Optional hooks​

HookDefault behavior
_reset_scene()Restores the initial state, then applies reset noise to pose and velocity.
_check_stop_criteria(action, observation, reward, info)Truncates the episode at steps_per_episode.
_reset_renderer()No-op; called before the first rendered frame after a reset.
_update_renderer()No-op; called before every rendered frame.
_init_ui()Adds the built-in Environment panel in the Polyscope viewer.

When overriding _reset_scene() or _check_stop_criteria(), call the base implementation to preserve state restoration or the episode-length limit. Call super()._check_stop_criteria(...) with keyword arguments because implementations may declare the parameters in a different order.

_apply_action runs once per simulation substep

Each control step runs simulation_frequency // control_frequency simulation substeps and calls _apply_action before every substep. Scale accumulated target changes by the simulation timestep, and reset Python-side accumulators in _reset_scene() after calling super().

Ending an episode​

  • Set self._terminated = True for task outcomes such as success or failure.
  • Set self._truncated = True for external limits. The base implementation already truncates at steps_per_episode.
  • Add info["terminated_reason"] or info["truncated_reason"] to explain why the episode ended.

Base configuration​

FieldTypeDefault
control_frequencyintrequired
simulation_frequencyintrequired
steps_per_episodeint-1 (no limit)
reset_noise_scalefloat0.0
num_worker_threadsint0
use_shared_scenesboolTrue
render_modestr | NoneNone
render_sizetuple[int, int] | NoneNone
render_coordinate_systemCoordinateSystem | str | NoneNone
start_pausedboolFalse
profileboolFalse
dump_timings_to_infoboolTrue

The constructor requires positive frequencies, requires the simulation frequency to be an integer multiple of the control frequency, and rejects negative reset noise. See Rendering for renderer-specific fields.

Pose reset noise is uniform and clipped to the articulation's DOF limits. Velocity noise is Gaussian and is not clipped. Both use the generator seeded by reset(seed=...).

Scene teardown callbacks

_load_scene(name, scene_builder) accepts (scene, agent) or (scene, agent, cleanup_callbacks). Use cleanup callbacks for resources that must be destroyed before their scene.

Registration​

Environments are registered explicitly -- SuperDex does not scan the filesystem for *_env.py modules. The shipped public environments are declared as a list of EnvSpecs in superdex.lab.gym.registration and registered by register_envs(); importing superdex.lab.gym calls that for you. (Internal environments are declared in superdex.lab.gym.internal.registration and registered by register_internal_envs().)

Registering an environment​

Declare a spec, register it, then gym.make its canonical ID:

import gymnasium as gym
from gymnasium.envs.registration import EnvSpec
from superdex.lab.gym.registration import register_env_spec

register_env_spec(
EnvSpec(
id="superdex_gym/MyCustom-v0",
entry_point="superdex.lab.gym.registration:make_superdex_env",
kwargs={
"env_cls": "my_package.my_custom_env:MyCustomEnv",
"cfg": {},
},
)
)
env = gym.make("superdex_gym/MyCustom-v0", cfg={"use_gravity": False})

Each spec carries:

  • id -- the canonical Gymnasium ID in the superdex_gym namespace, ending in -vN. This is exactly what gym.make and the CLI accept; there are no short names or discovery aliases.
  • entry_point -- always the shared make_superdex_env factory ("superdex.lab.gym.registration:make_superdex_env"), which loads env_cls and merges any gym.make keyword overrides into cfg.
  • kwargs["env_cls"] -- "<module>:<ClassName>" for the concrete MochiEnv subclass.
  • kwargs["cfg"] -- the registered configuration. Overrides passed to gym.make(id, cfg={...}) are deep-merged over this registered cfg.

register_env_spec registers the spec; get_env_specs() simply scans the Gymnasium registry for the superdex_gym namespace, so any spec registered this way appears in the catalog helpers (there is no separate "recording" step). Registration is idempotent for an equivalent spec -- a matching ID already in the registry is left alone, while an incompatible existing registration raises ValueError.

Names are canonical IDs​

ClassCanonical Gymnasium ID
CartPoleEnvsuperdex_gym/CartPole-v0
HalfCheetahEnvsuperdex_gym/HalfCheetah-v0
MyCustomEnvsuperdex_gym/MyCustom-v0

The id is chosen when you declare the spec, not derived from the module or class name. By convention it drops the Env suffix and keeps PascalCase. The canonical ID is the only public name -- gym.make and the CLI accept it directly, with no short-name aliases.

Listing registered environments​

from superdex.lab.gym.registration import get_env_specs, register_envs

register_envs()
for spec in get_env_specs():
print(spec.id)

Config variants are additional specs​

A configuration variant is simply another EnvSpec that reuses the same env_cls with a different registered cfg and its own canonical ID -- not a separate JSON file or a discovered artifact. The shipped CartPole, for example, declares a base and a variant:

EnvSpec(
id="superdex_gym/CartPole-v0",
entry_point="superdex.lab.gym.registration:make_superdex_env",
kwargs={
"env_cls": "superdex.lab.gym.envs.benchmarks.cartpole_env:CartPoleEnv",
"cfg": {},
},
)
EnvSpec(
id="superdex_gym/CartPoleActuateOnPole-v0",
entry_point="superdex.lab.gym.registration:make_superdex_env",
kwargs={
"env_cls": "superdex.lab.gym.envs.benchmarks.cartpole_env:CartPoleEnv",
"cfg": {"actuate_on_pole": True},
},
)

Every trainable configuration is its own registered canonical ID, which is what lets the RLlib recipes below map one-to-one to environments.

Test-only environments​

Environments that exist only to be crash-checked -- degenerate configurations not worth shipping as tasks -- are not registered at import time. Register them from within the tests that exercise them, so they never enter the shipped registry or the catalog helpers.

RLlib recipes use explicit manifests​

RLlib recipes are separate from environment discovery. Public recipes live under superdex/lab/rllib/recipes/. That directory has a manifest.json that maps an exact canonical Gymnasium ID to a filesystem-safe output slug and explicit train or benchmark recipe paths. The canonical loader is superdex.lab.rllib.recipe_manifest.

A variant recipe does not fall back to the base recipe: only IDs listed explicitly in a manifest have recipes, and a base does not inherit a variant's recipe either. Training recipes contain algorithm settings, not environment configuration, so every trained configuration must be represented by its own registered canonical Gymnasium ID.

Scene and Actor Helpers​

superdex.lab.gym.utils.mochi_helpers holds the utilities the shipped environments use to build scenes and read agent state.

Building a scene directly​

When a prefab is not appropriate, construct the scene and articulated agent in the builder. Import superdex.physics as physics, and replace the placeholder actor parameters before running this version:

    def _init_scene(self, cfg: MyCustomEnvCfg):
"""Create or join the shared scene, then capture the initial agent state."""

def scene_builder():
scene = physics.create_scene("MyCustomScene")

if not cfg.use_gravity:
scene.set_gravity([0, 0, 0])

agent_params = physics.ArticulatedActorParams()
# Configure agent_params here...
agent = scene.create_articulated_actor(agent_params)

return scene, agent

# Include every field that changes scene construction. Matching environments
# share a scene within the process.
uid_fields = (cfg.use_gravity,)
self._load_scene(f"scene_{hash(uid_fields)}", scene_builder)

# Required by the default _reset_scene().
self._initial_pose = mochi_helpers.get_articulated_pose(self._agent)
self._initial_velocity = mochi_helpers.get_articulated_joint_velocities(
self._agent
)

For an agent with a free-floating root, derive the controlled size and apply forces to all non-root DOFs:

# Derive space sizes from the agent that was just loaded.
num_dofs = self._agent.get_num_dofs()
num_controlled_dofs = num_dofs - 6 # This agent has a free-floating root.

self._setup_action_space(
control=ActionSpace(-1.0, 1.0, (num_controlled_dofs,), dtype=np.float32),
)
# Force control: apply generalized forces to the controlled DOFs.
force_scale = 100.0
dof_indices = list(range(6, 6 + control.shape[0]))
self._agent.set_external_forces_on_dofs(dof_indices, control * force_scale)

Building a scene from a prefab​

init_prefab_scene(prefab, params) -> (scene, agent) is the preferred helper, and all shipped environments use it. It creates a scene from a ScenePrefab or prefab file path, optionally adds a ground plane, and selects the agent actor. The complete workflow above shows the pattern, including a builder closure that reads cfg and a shared-scene key that includes every field affecting scene construction.

PrefabParams adds these fields to the base SuperDex Physics prefab parameters:

FieldDefaultMeaning
root_dir""Where relative paths inside the prefab resolve from. Empty means the SuperDex Physics assets directory.
agent_actor_name""Which actor becomes self._agent. Empty auto-selects the scene's single articulated actor. Names may contain slashes when they come from nested prefabs.
add_ground_planeTrueAdd a ground plane
ground_normal(0, 1, 0)Ground plane normal
ground_offset0.0Plane offset along ground_normal
ground_contact_paramsNoneContact parameters for the ground plane; None uses SuperDex Physics defaults

It also inherits name, scale, rotation, translation and apply_scene_settings from the base class.

Auto-selection requires exactly one articulated actor. Set agent_actor_name explicitly when the scene contains zero or multiple articulated actors.

Reading agent state​

FunctionReturns
get_articulated_pose(actor)Generalized pose, one element per DOF
get_articulated_joint_velocities(actor)Generalized velocity, one element per DOF
get_articulated_dof_limits(actor)Per-DOF (min, max) limits
get_contact_force_and_torque_world(actor)The 6-D contact wrench on a rigid actor
get_actors(scene)Every actor in the scene
TransformRT_to_numpy(transform)A (2, 3) array containing translation and rotation-vector rows

get_articulated_dof_limits is what the default _reset_scene() uses to clip the pose noise, and it is the right source for validating your own joint targets.