From 672f77d4ed4da1f5540ad0f11327dd0d72f34f59 Mon Sep 17 00:00:00 2001
From: YUFEI JIA <59379871+TATP-233@users.noreply.github.com>
Date: Sat, 5 Sep 2026 22:38:34 +0800
Subject: [PATCH 1/2] fix: route GPU backend environments by data-parallel rank
---
.../en/2-user_guide/2-algorithms/1-ppo.md | 7 +
.../en/2-user_guide/2-algorithms/3-sac.md | 5 +
.../zh_CN/2-user_guide/2-algorithms/1-ppo.md | 5 +
.../zh_CN/2-user_guide/2-algorithms/3-sac.md | 2 +
src/unilab/base/backend_factory.py | 59 +++-
src/unilab/base/base.py | 13 +
src/unilab/base/env_factory.py | 21 ++
src/unilab/base/process_device.py | 260 +++++++++++++++++-
.../conf/ppo/task/g1_walk_flat/genesis.yaml | 4 +
.../conf/sac/task/g1_walk_flat/genesis.yaml | 4 +
src/unilab/scripts/play_hora_appo.py | 29 +-
src/unilab/scripts/play_interactive.py | 29 +-
src/unilab/scripts/train_appo.py | 51 +++-
src/unilab/scripts/train_offpolicy.py | 108 +++++++-
src/unilab/scripts/train_rsl_rl.py | 113 +++++++-
tests/base/backend/test_process_device.py | 78 ++++++
tests/base/test_genesis_backend.py | 6 +-
tests/envs/test_env_configs.py | 1 +
tests/scripts/test_train_scripts.py | 46 ++++
19 files changed, 800 insertions(+), 41 deletions(-)
diff --git a/docs/sphinx/source/en/2-user_guide/2-algorithms/1-ppo.md b/docs/sphinx/source/en/2-user_guide/2-algorithms/1-ppo.md
index 4277eff75..358f813bc 100644
--- a/docs/sphinx/source/en/2-user_guide/2-algorithms/1-ppo.md
+++ b/docs/sphinx/source/en/2-user_guide/2-algorithms/1-ppo.md
@@ -50,6 +50,13 @@ device. Do not set `training.device` and `training.devices` together. The
configured order is preserved, including when the parent already has
`CUDA_VISIBLE_DEVICES` set.
+For the IsaacGym, IsaacSim, and Genesis owners, the same topology is also
+applied to the simulator environment. Torchrun workers receive the local
+index inside their remapped `CUDA_VISIBLE_DEVICES` list (for example, host
+device 5 is sent as `device_id=1` when the worker sees `[4,5]`); off-policy
+workers keep the parent process's visible index namespace. Genesis selects
+its process-wide session before `gs.init`.
+
`algo.num_envs` is a **per-rank** count, not a global budget. For `W` ranks,
`N` configured envs, and rollout length `T`:
diff --git a/docs/sphinx/source/en/2-user_guide/2-algorithms/3-sac.md b/docs/sphinx/source/en/2-user_guide/2-algorithms/3-sac.md
index d45cdf62f..1d945c93f 100644
--- a/docs/sphinx/source/en/2-user_guide/2-algorithms/3-sac.md
+++ b/docs/sphinx/source/en/2-user_guide/2-algorithms/3-sac.md
@@ -54,6 +54,11 @@ materialization. The collector therefore does not fall back to Warp's fresh-proc
of `cuda:0`. The local binding is recorded as `collector_backend_device` in the runtime
manifest.
+IsaacGym, IsaacSim, and Genesis receive the rank-selected simulator device
+through the environment override as well. Off-policy collectors use the
+parent's visible CUDA indices; Genesis binds its process-wide session before
+initialization.
+
MuJoCo has a committed multi-GPU scaling benchmark. The mjwarp per-rank placement contract is
covered by `tests/base/backend/test_process_device.py` and the off-policy runner/worker unit
tests; the repository does not currently contain an mjwarp multi-GPU throughput or convergence
diff --git a/docs/sphinx/source/zh_CN/2-user_guide/2-algorithms/1-ppo.md b/docs/sphinx/source/zh_CN/2-user_guide/2-algorithms/1-ppo.md
index 26094744c..718f33115 100644
--- a/docs/sphinx/source/zh_CN/2-user_guide/2-algorithms/1-ppo.md
+++ b/docs/sphinx/source/zh_CN/2-user_guide/2-algorithms/1-ppo.md
@@ -48,6 +48,11 @@ uv run train --algo ppo --task g1_motion_tracking --sim mujoco \
`training.devices`。父进程已有 `CUDA_VISIBLE_DEVICES` 时,配置索引仍按父进程可见
设备解释,并保留用户给定顺序。
+对 IsaacGym、IsaacSim 和 Genesis owner,同一拓扑也会传给环境仿真器。torchrun
+worker 继承重映射后的 `CUDA_VISIBLE_DEVICES`,因此传给 worker 的是本地索引(例如
+worker 看到 `[4,5]` 时,主机设备 5 传为 `device_id=1`);off-policy worker 保持父进程
+可见设备索引。Genesis 会在 `gs.init` 前选择每个进程的 session 设备。
+
`algo.num_envs` 是**每个 rank** 的环境数,不是全局预算。设 rank 数为 `W`、配置
环境数为 `N`、rollout 长度为 `T`:
diff --git a/docs/sphinx/source/zh_CN/2-user_guide/2-algorithms/3-sac.md b/docs/sphinx/source/zh_CN/2-user_guide/2-algorithms/3-sac.md
index c0e2705b5..734044002 100644
--- a/docs/sphinx/source/zh_CN/2-user_guide/2-algorithms/3-sac.md
+++ b/docs/sphinx/source/zh_CN/2-user_guide/2-algorithms/3-sac.md
@@ -97,6 +97,8 @@ MuJoCo worker 线程逐核绑定外,collector 进程本身(含 Numba 并行
- MuJoCo 有已提交的多卡 scaling benchmark;mjwarp 的 per-rank device placement 有
`tests/base/backend/test_process_device.py` 与 off-policy runner/worker 单测覆盖,但仓库中
尚无 mjwarp 多卡吞吐或收敛 benchmark。
+- IsaacGym、IsaacSim 和 Genesis 的 collector 环境也会收到 rank 对应的仿真设备。off-policy
+ 使用父进程可见的 CUDA 索引;Genesis 在初始化 `gs.init` 前绑定进程级 session 设备。
- 仅单节点:rank 之间通过 run 目录里的 FileStore rendezvous,NCCL 走 TCP
loopback(默认 `NCCL_P2P_DISABLE=1` / `NCCL_SHM_DISABLE=1`,环境变量显式设置
时优先)——部分机型(如 RTX 6000D)的 NCCL P2P/SHM peer transport 不可靠,
diff --git a/src/unilab/base/backend_factory.py b/src/unilab/base/backend_factory.py
index e9bca6e02..e88356622 100644
--- a/src/unilab/base/backend_factory.py
+++ b/src/unilab/base/backend_factory.py
@@ -14,15 +14,35 @@
from unisim.backend.base import SimBackend
from unilab.assets.hub import ensure_robot_assets_for_paths
+from unilab.base.process_device import bind_genesis_process_device
if TYPE_CHECKING:
from unilab.base.base import EnvCfg
from unilab.base.scene import SceneCfg
+def _legacy_genesis_device_option_error(exc: TypeError) -> bool:
+ """Identify an old UniSim adapter rejecting the optional device keyword.
+
+ UniSim 1.1 reports unknown backend options from ``GenesisBackend`` while
+ other compatible releases may expose Python's usual ``unexpected keyword``
+ wording. Keep the compatibility retry narrowly scoped to those messages;
+ constructor errors from the actual Genesis runtime must still propagate.
+ """
+
+ message = str(exc).lower()
+ mentions_device = "genesis_device_id" in message or "device_id" in message
+ rejects_keyword = (
+ "does not accept backend options" in message
+ or "unexpected keyword argument" in message
+ or "unexpected keyword" in message
+ )
+ return mentions_device and rejects_keyword
+
+
def env_backend_kwargs(cfg: "EnvCfg") -> dict[str, Any]:
"""Translate ``EnvCfg`` backend knobs into UniSim adapter options."""
- return {
+ result: dict[str, Any] = {
"post_step_forward_sensor": cfg.post_step_forward_sensor,
"motrix_max_iterations": cfg.motrix_max_iterations,
"chunk_size": cfg.chunk_size,
@@ -45,6 +65,13 @@ def env_backend_kwargs(cfg: "EnvCfg") -> dict[str, Any]:
"isaacsim_render_width": cfg.isaacsim_render_width,
"isaacsim_render_height": cfg.isaacsim_render_height,
}
+ # Keep the optional key absent for legacy unisim-core releases that do not
+ # know about Genesis' explicit device argument. Once a rank selects a
+ # device the key is added below and ``create_backend`` supplies a narrow
+ # compatibility fallback for those releases.
+ if cfg.genesis_device_id is not None:
+ result["genesis_device_id"] = cfg.genesis_device_id
+ return result
def create_backend(
@@ -63,7 +90,35 @@ def create_backend(
[scene.model_file, scene.visual_model_file, *scene.fragment_files]
)
kwargs["body_state_required"] = body_state_required
- return unisim.create_backend(backend_type, scene, num_envs, sim_dt, **kwargs)
+ if backend_type == "genesis" and kwargs.get("genesis_device_id") is not None:
+ # Bind before any unisim-core Genesis constructor can call gs.init.
+ # New unisim-core releases repeat this idempotently; old releases do
+ # not accept the keyword, so the retry below still gets the correct
+ # process-wide current device.
+ genesis_device_id = kwargs["genesis_device_id"]
+ if (
+ isinstance(genesis_device_id, bool)
+ or not isinstance(genesis_device_id, int)
+ or genesis_device_id < 0
+ ):
+ raise ValueError(
+ "genesis_device_id must be a non-negative integer or None, "
+ f"got {genesis_device_id!r}"
+ )
+ bind_genesis_process_device(f"cuda:{genesis_device_id}")
+ try:
+ return unisim.create_backend(backend_type, scene, num_envs, sim_dt, **kwargs)
+ except TypeError as exc:
+ if backend_type != "genesis" or "genesis_device_id" not in kwargs:
+ raise
+ # unisim-core < 1.2 has no Genesis device field and reports the
+ # unknown option from GenesisBackend. Retry only for that precise
+ # capability error; unrelated constructor TypeErrors must propagate.
+ if not _legacy_genesis_device_option_error(exc):
+ raise
+ legacy_kwargs = dict(kwargs)
+ legacy_kwargs.pop("genesis_device_id", None)
+ return unisim.create_backend(backend_type, scene, num_envs, sim_dt, **legacy_kwargs)
__all__ = ["SimBackend", "create_backend", "env_backend_kwargs"]
diff --git a/src/unilab/base/base.py b/src/unilab/base/base.py
index 59b6dd82e..8f386d291 100644
--- a/src/unilab/base/base.py
+++ b/src/unilab/base/base.py
@@ -56,6 +56,10 @@ class EnvCfg:
# backend defaults (device 0, generous handshake/step timeout).
isaacgym_device_id: Optional[int] = None
isaacgym_worker_timeout_s: Optional[float] = None
+ # ``genesis`` owns one process-wide GPU session. The explicit device id
+ # must be selected before ``gs.init`` so each data-parallel rank gets its
+ # own simulator device; ``None`` keeps Genesis' current/default device.
+ genesis_device_id: Optional[int] = None
# ``genesis`` drops the MJCF global