From 651f5f93600309eaf47b7e78682b113703510f28 Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Wed, 17 Jun 2026 11:55:29 -0500 Subject: [PATCH 01/45] feat(dreamzero): add launcher scripts + 14B eval config --- .../libero/scripts/convert_checkpoint.sh | 56 +++++ .../libero_spatial_eval_dreamzero_14b.yaml | 235 ++++++++++++++++++ .../libero/scripts/run_dreamzero_eval_eks.sh | 93 +++++++ .../libero/scripts/run_dreamzero_sft_eks.sh | 156 ++++++++++++ 4 files changed, 540 insertions(+) create mode 100755 3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/convert_checkpoint.sh create mode 100644 3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/libero_spatial_eval_dreamzero_14b.yaml create mode 100755 3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/run_dreamzero_eval_eks.sh create mode 100755 3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/run_dreamzero_sft_eks.sh diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/convert_checkpoint.sh b/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/convert_checkpoint.sh new file mode 100755 index 000000000..9d1b9d7f8 --- /dev/null +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/convert_checkpoint.sh @@ -0,0 +1,56 @@ +#!/bin/bash +# ============================================================================= +# DreamZero checkpoint conversion: FSDP DCP shards -> single .pt +# +# RLinf SFT saves sharded FSDP2 checkpoints under +# {log_path}/{experiment_name}/checkpoints/global_step_/actor/dcp_checkpoint/ +# as ___0.distcp files. The LIBERO simulator eval (eval_embodied_agent.py) +# consumes a single consolidated .pt via runner.ckpt_path. This script converts +# the DCP shards to that .pt using upstream's convert_dcp_to_pt.py. +# +# Env: +# DCP_PATH - dcp_checkpoint dir (default derived from CKPT_PATH/EXPERIMENT_NAME/STEP) +# OUTPUT_PT - output .pt path (default: /actor/model_state_dict/full_weights.pt) +# CKPT_PATH, EXPERIMENT_NAME, STEP - used to derive defaults +# ============================================================================= +set -euo pipefail +set -x + +VENV_NAME="${VENV_NAME:-dreamzero}" +CKPT_PATH="${CKPT_PATH:-/fsx/checkpoints}" +EXPERIMENT_NAME="${EXPERIMENT_NAME:-dreamzero-libero-sft}" +# Upstream nests under the config's runner.logger.experiment_name (libero_sft_dreamzero). +INNER_EXPERIMENT="${INNER_EXPERIMENT:-libero_sft_dreamzero}" +STEP="${STEP:-global_step_1}" + +CKPT_DIR="${CKPT_PATH}/${EXPERIMENT_NAME}/${INNER_EXPERIMENT}/checkpoints/${STEP}" +DCP_PATH="${DCP_PATH:-${CKPT_DIR}/actor/dcp_checkpoint}" +OUTPUT_PT="${OUTPUT_PT:-${CKPT_DIR}/actor/model_state_dict/full_weights.pt}" + +export PYTHONPATH="${PYTHONPATH:-}" +UV_PATH="${UV_PATH:-/opt/venv}" +if [ -f "${UV_PATH}/${VENV_NAME}/bin/activate" ]; then + source "${UV_PATH}/${VENV_NAME}/bin/activate" +fi +export DREAMZERO_PATH="${DREAMZERO_PATH:-/workspace/DreamZero}" +export PYTHONPATH="${DREAMZERO_PATH}:${PYTHONPATH}" + +cd /workspace/RLinf + +if [ ! -d "${DCP_PATH}" ]; then + echo "ERROR: DCP checkpoint dir not found: ${DCP_PATH}" + exit 1 +fi + +mkdir -p "$(dirname "${OUTPUT_PT}")" + +echo "=== Converting DCP -> .pt ===" +echo "DCP: ${DCP_PATH}" +echo "Output: ${OUTPUT_PT}" + +python3 rlinf/utils/ckpt_convertor/fsdp_convertor/convert_dcp_to_pt.py \ + --dcp_path "${DCP_PATH}" \ + --output_path "${OUTPUT_PT}" + +echo "=== Done ===" +ls -la "${OUTPUT_PT}" diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/libero_spatial_eval_dreamzero_14b.yaml b/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/libero_spatial_eval_dreamzero_14b.yaml new file mode 100644 index 000000000..31c8942a6 --- /dev/null +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/libero_spatial_eval_dreamzero_14b.yaml @@ -0,0 +1,235 @@ +# libero_spatial_eval_dreamzero_14b.yaml +# ============================================================================= +# DreamZero 14B LIBERO-Spatial simulator eval config (RLinf-on-EKS variant). +# +# WHY THIS FILE EXISTS: +# Upstream RLinf only ships a *5B* embodied eval config +# (examples/embodiment/config/libero_spatial_eval_dreamzero.yaml), which pins +# the 5B WAM architecture (`model/dreamzero_5b@actor.model`: dim 3072, 30 +# layers) and Wan2.2-TI2V-5B component paths. OUR trained checkpoint is the +# 14B WAM (dim 5120, 40 layers), warm-started from DreamZero-DROID and SFT'd on +# LIBERO (examples/sft/config/libero_sft_dreamzero_14b.yaml). Evaluating a 14B +# full_weights.pt against a 5B model architecture fails to load +# (HuggingFaceWorker.init_worker -> model.load_state_dict(model_dict) is a +# strict load; the parameter shapes will not match). This config selects the +# 14B architecture instead. +# +# HOW THE 14B ARCHITECTURE RESOLVES: +# The embodied eval entrypoint (examples/embodiment/eval_embodied_agent.py) +# runs with `--config-path config`, i.e. examples/embodiment/config/. That +# directory ships ONLY model/dreamzero_5b.yaml -- there is no +# model/dreamzero_14b.yaml there. The 14B architecture lives in the SFT config +# tree at examples/sft/config/model/dreamzero_14b.yaml. We therefore add the +# SFT config dir to the Hydra `searchpath` below so the +# `model/dreamzero_14b@actor.model` default resolves at eval time. The path is +# the deterministic in-container location (RLinf/ is COPYed to /workspace/RLinf +# by examples/Dockerfile). +# +# Unlike the SFT path (validate_dreamzero_sft_model_cfg merges the checkpoint's +# config.json into actor.model), the EMBODIED eval path (validate_embodied_cfg) +# does NOT merge config.json. The architecture comes entirely from this YAML, +# so the `model/dreamzero_14b` default must be 14B-correct on its own. +# +# HOW THIS FILE IS DELIVERED TO THE CONTAINER: +# This file lives in OUR repo (examples/dreamzero/configs/), NOT upstream. At +# runtime it is mounted into the eval pod via the `dreamzero-eval-config` +# ConfigMap at /opt/eval-config/, and run_dreamzero_eval_eks.sh copies it into +# /workspace/RLinf/examples/embodiment/config/ before invoking the eval (so +# `--config-path config` finds it WITHOUT hiding the upstream env/ and model/ +# config groups it depends on). See examples/dreamzero/manifests/dreamzero-eval.yaml. +# +# DROID action space: 8D (7 joint + gripper); model outputs action chunks. +# ============================================================================= + +defaults: + - env/libero_spatial@env.train + - env/libero_spatial@env.eval + # 14B WAM architecture (NOT the 5B model/dreamzero_5b). Resolved from the SFT + # config tree via the hydra.searchpath entry below. + - model/dreamzero_14b@actor.model + - training_backend/fsdp@actor.fsdp_config + - weight_syncer/patch_syncer@weight_syncer + - override hydra/job_logging: stdout + +hydra: + run: + dir: . + output_subdir: null + searchpath: + - file://${oc.env:EMBODIED_PATH}/config/ + # The 14B model arch (model/dreamzero_14b.yaml) ships only under the SFT + # config tree, not the embodiment one. Add it to the searchpath so the + # `model/dreamzero_14b@actor.model` default above resolves. This is the + # deterministic in-container path (examples/Dockerfile COPYs RLinf/ here). + - file:///workspace/RLinf/examples/sft/config/ + +cluster: + num_nodes: 1 + component_placement: + actor,env,rollout: all + +runner: + task_type: embodied + logger: + log_path: "../results" + project_name: rlinf + experiment_name: "libero_spatial_eval_dreamzero_14b" + logger_backends: ["tensorboard"] + + max_epochs: 1 + only_eval: True + val_check_interval: -1 + save_interval: -1 + resume_dir: null + ckpt_path: /path/to/model.pt + +algorithm: + normalize_advantages: True + kl_penalty: kl + group_size: 4 + rollout_epoch: 1 + eval_rollout_epoch: 1 + + reward_type: chunk_level + logprob_type: chunk_level + entropy_type: chunk_level + + update_epoch: 5 + adv_type: grpo + loss_type: actor_critic + + gamma: 0.99 + gae_lambda: 0.95 + bootstrap_type: always + kl_beta: 0.0 + entropy_bonus: 0.005 + clip_ratio_high: 0.2 + clip_ratio_low: 0.2 + clip_ratio_c: 3.0 + value_clip: 0.2 + huber_delta: 10.0 + + length_params: + max_new_token: null + max_length: 1024 + min_length: 1 + + sampling_params: + do_sample: True + temperature_train: 1.0 + temperature_eval: 0.6 + top_k: 50 + top_p: 1.0 + repetition_penalty: 1.0 + +env: + group_name: "EnvGroup" + + train: + total_num_envs: 8 + max_episode_steps: 480 + max_steps_per_rollout_epoch: 480 + + eval: + # LIBERO-Spatial: 10 tasks x 50 trials = 500 init states. With total_num_envs parallel envs, + # set max_steps_per_rollout_epoch to ceil(500/total_num_envs) * max_episode_steps so auto_reset + # cycles through the full suite within one eval_rollout_epoch (see vla-eval.rst). + # + # GPU-MEMORY NOTE (14B): the upstream 5B eval uses total_num_envs=128, but the + # 16.48B DreamZero model + LIBERO sim co-located on one node's 8x H200 (139GB + # each) OOMs at 128 envs (only ~280MB free on GPU 0). 16 envs fits comfortably. + # Increase only if you add nodes or reduce per-GPU model memory. For a quick + # smoke eval, override to 8 via HYDRA_OVERRIDES. + total_num_envs: 16 + auto_reset: True + ignore_terminations: True + max_episode_steps: 480 + max_steps_per_rollout_epoch: 960 # ceil(500/16) * ... -> covers the suite; raise for full coverage + group_size: 1 + use_fixed_reset_state_ids: True + use_ordered_reset_state_ids: True + is_eval: True + + video_cfg: + save_video: True + video_base_dir: ${runner.logger.log_path}/video/eval + +rollout: + group_name: "RolloutGroup" + generation_backend: "huggingface" + enable_offload: False + pipeline_stage_num: 1 + + model: + model_path: ${actor.model.model_path} + precision: ${actor.model.precision} + +actor: + group_name: "ActorGroup" + training_backend: "fsdp" + micro_batch_size: 1 + global_batch_size: 64 + seed: 0 + enable_offload: False + + model: + model_type: "dreamzero" + precision: bf16 + # model_path is set to /fsx/models/DreamZero-DROID by the launcher + # (actor.model.model_path=${MODEL_PATH}). The DreamZero-DROID safetensors + # provide the 14B backbone init; our trained full_weights.pt (runner.ckpt_path) + # is then strict-loaded over it. Leave null here so the launcher controls it. + model_path: null + # umt5-xxl tokenizer. Launcher overrides to /fsx/models/umt5-xxl. + tokenizer_path: google/umt5-xxl # https://huggingface.co/google/umt5-xxl + + # NOTE (5B vs 14B): the upstream 5B eval config OVERRIDES the component + # pretrained paths here to Wan2.2-TI2V-5B (diffusion/image/text/vae). We do + # NOT do that for 14B. model/dreamzero_14b.yaml already sets all four + # component paths to null, and skip_component_loading is enabled at build + # time when full DreamZero safetensors are present under model_path + # (rlinf/models/embodiment/dreamzero/__init__.py). The 14B backbone weights + # come from the DreamZero-DROID safetensors + our full_weights.pt, NOT from a + # 5B Wan2.2-TI2V-5B download. Re-specifying 5B paths here would force a wrong + # (5B) backbone build, so they are intentionally omitted (left null). + + # Dataset-generated metadata.json for normalisation statistics (libero_sim). + # Generate with: python toolkits/lerobot/generate_dreamzero_metadata.py \ + # --preset libero_sim --dataset-root /path/to/libero --output-metadata /path/to/metadata.json + # IMPORTANT: keep this COMMENTED OUT. model/dreamzero_14b.yaml does not define + # metadata_json_path, so the launcher ADDS it with the Hydra '+' prefix + # (+actor.model.metadata_json_path=${METADATA_PATH}). Defining it here would + # make the '+' append fail ("key already exists"). + # metadata_json_path: /path/to/metadata.json + + embodiment_tag: "libero_sim" + + is_lora: False + # LIBERO temporal alignment: action_horizon, num_action_chunks and + # num_action_per_block must all be 16 (same fix as the LIBERO SFT path). + # model/dreamzero_14b.yaml defaults num_action_per_block to 24 (a DROID + # value); overriding to 16 keeps num_action_per_block // num_frame_per_block + # consistent with the LIBERO action layout (see run_dreamzero_sft_eks.sh). + action_horizon: 16 + num_action_chunks: 16 + num_action_per_block: 16 + # 14B render size (model/dreamzero_14b.yaml: 352x640). This differs from the + # 5B eval config, which uses 160x320. The video tile/stride geometry of the + # 14B DiT expects the larger frame. + target_video_height: 352 + target_video_width: 640 + + + fsdp_config: + strategy: "fsdp" + gradient_checkpointing: False + mixed_precision: + param_dtype: ${actor.model.precision} + reduce_dtype: ${actor.model.precision} + buffer_dtype: ${actor.model.precision} + +reward: + use_reward_model: False + +critic: + use_critic_model: False diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/run_dreamzero_eval_eks.sh b/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/run_dreamzero_eval_eks.sh new file mode 100755 index 000000000..9c85779ae --- /dev/null +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/run_dreamzero_eval_eks.sh @@ -0,0 +1,93 @@ +#!/bin/bash +# ============================================================================= +# DreamZero LIBERO simulator eval launcher (single-node, GPU). +# +# Runs upstream examples/embodiment/eval_embodied_agent.py against a trained +# DreamZero checkpoint (full_weights.pt) in the LIBERO Spatial simulator, +# reporting success_once and (optionally) saving in-sim rollout videos. +# +# Env: +# VENV_NAME - venv (default: dreamzero) +# CONFIG_NAME - eval config (default: libero_spatial_eval_dreamzero_14b) +# CKPT_PT - full_weights.pt to evaluate (runner.ckpt_path) +# METADATA_PATH - libero_sim metadata.json (default: /fsx/models/metadata-libero.json) +# TOKENIZER_PATH - umt5-xxl (default: /fsx/models/umt5-xxl) +# MODEL_PATH - DreamZero-DROID backbone dir (for component build; default staged) +# LOG_DIR - eval output (default: /fsx/checkpoints/dreamzero-libero-eval) +# SAVE_VIDEO - "True"/"False" save in-sim rollout video (default: True) +# HYDRA_OVERRIDES - extra Hydra args +# ============================================================================= +set -euo pipefail +set -x + +VENV_NAME="${VENV_NAME:-dreamzero}" +CONFIG_NAME="${CONFIG_NAME:-libero_spatial_eval_dreamzero_14b}" +CKPT_PT="${CKPT_PT:?set CKPT_PT to the full_weights.pt to evaluate}" +METADATA_PATH="${METADATA_PATH:-/fsx/models/metadata-libero.json}" +TOKENIZER_PATH="${TOKENIZER_PATH:-/fsx/models/umt5-xxl}" +MODEL_PATH="${MODEL_PATH:-/fsx/models/DreamZero-DROID}" +LOG_DIR="${LOG_DIR:-/fsx/checkpoints/dreamzero-libero-eval}" +SAVE_VIDEO="${SAVE_VIDEO:-True}" + +export PYTHONPATH="${PYTHONPATH:-}" +UV_PATH="${UV_PATH:-/opt/venv}" +if [ -f "${UV_PATH}/${VENV_NAME}/bin/activate" ]; then + echo "Activating venv: ${VENV_NAME}" + source "${UV_PATH}/${VENV_NAME}/bin/activate" +elif [ -f "/usr/local/bin/switch_env" ]; then + source switch_env "${VENV_NAME}" +else + echo "WARNING: No venv found for ${VENV_NAME}, using system Python" +fi + +export DREAMZERO_PATH="${DREAMZERO_PATH:-/workspace/DreamZero}" +export PYTHONPATH="${DREAMZERO_PATH}:${PYTHONPATH}" +export EMBODIED_PATH="/workspace/RLinf/examples/embodiment" +# LIBERO simulator requires headless GL; osmesa is the safe EKS default. +export MUJOCO_GL="${MUJOCO_GL:-osmesa}" +export PYOPENGL_PLATFORM="${PYOPENGL_PLATFORM:-osmesa}" + +cd /workspace/RLinf + +# --- Stage the 14B eval config into the embodiment config dir --- +# eval_embodied_agent.py runs with `--config-path config`, i.e. +# /workspace/RLinf/examples/embodiment/config/. Upstream only ships a 5B eval +# config there; our 14B variant is mounted (via the dreamzero-eval-config +# ConfigMap) at /opt/eval-config/ and copied into the embodiment config dir here. +# We copy a single file (NOT mount a ConfigMap over the dir) so the upstream +# config groups (env/, model/, training_backend/, weight_syncer/) remain visible. +EVAL_CONFIG_SRC="${EVAL_CONFIG_SRC:-/opt/eval-config}" +EMBODIED_CONFIG_DIR="/workspace/RLinf/examples/embodiment/config" +if [ -f "${EVAL_CONFIG_SRC}/${CONFIG_NAME}.yaml" ]; then + echo "Staging eval config: ${EVAL_CONFIG_SRC}/${CONFIG_NAME}.yaml -> ${EMBODIED_CONFIG_DIR}/" + cp "${EVAL_CONFIG_SRC}/${CONFIG_NAME}.yaml" "${EMBODIED_CONFIG_DIR}/${CONFIG_NAME}.yaml" +else + echo "No mounted eval config at ${EVAL_CONFIG_SRC}/${CONFIG_NAME}.yaml;" + echo "expecting ${EMBODIED_CONFIG_DIR}/${CONFIG_NAME}.yaml to already exist." +fi + +if [ ! -f "${CKPT_PT}" ]; then + echo "ERROR: checkpoint not found: ${CKPT_PT}" + echo "Convert the DCP checkpoint to .pt first (convert-checkpoint.yaml)." + exit 1 +fi +mkdir -p "${LOG_DIR}" + +HYDRA_ARGS="runner.only_eval=True" +HYDRA_ARGS="${HYDRA_ARGS} runner.ckpt_path=${CKPT_PT}" +HYDRA_ARGS="${HYDRA_ARGS} runner.logger.log_path=${LOG_DIR}" +HYDRA_ARGS="${HYDRA_ARGS} actor.model.tokenizer_path=${TOKENIZER_PATH}" +HYDRA_ARGS="${HYDRA_ARGS} actor.model.model_path=${MODEL_PATH}" +# metadata_json_path is commented out in the config struct -> add with '+'. +HYDRA_ARGS="${HYDRA_ARGS} +actor.model.metadata_json_path=${METADATA_PATH}" +HYDRA_ARGS="${HYDRA_ARGS} env.eval.video_cfg.save_video=${SAVE_VIDEO}" +if [ -n "${HYDRA_OVERRIDES:-}" ]; then + HYDRA_ARGS="${HYDRA_ARGS} ${HYDRA_OVERRIDES}" +fi + +echo "Eval Hydra args: ${HYDRA_ARGS}" +# shellcheck disable=SC2086 +python3 examples/embodiment/eval_embodied_agent.py \ + --config-path config \ + --config-name "${CONFIG_NAME}" \ + ${HYDRA_ARGS} diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/run_dreamzero_sft_eks.sh b/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/run_dreamzero_sft_eks.sh new file mode 100755 index 000000000..c9d0254bd --- /dev/null +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/run_dreamzero_sft_eks.sh @@ -0,0 +1,156 @@ +#!/bin/bash +# ============================================================================= +# DreamZero LIBERO SFT Launch Script for Amazon EKS (Ray / FSDP2) +# +# Launches supervised fine-tuning of DreamZero (14B WAM) on LIBERO using +# upstream RLinf's new SFT mechanism: the entrypoint examples/sft/train_vla_sft.py +# driven by RLinf's Cluster (Ray) scheduler with FSDP2. +# +# This script runs on the Ray HEAD pod AFTER the Ray workers have joined. +# RLinf's Cluster scheduler fans the training out across all nodes/GPUs in the +# Ray cluster, so this launcher does NOT use torchrun -- it simply invokes the +# Hydra-driven entrypoint once on the head. +# +# Key environment variables (set via K8s manifest env): +# VENV_NAME - Python venv to activate (default: dreamzero) +# CONFIG_NAME - Hydra config name (default: libero_sft_dreamzero_14b) +# MODEL_PATH - 14B warm-start checkpoint (default: /fsx/models/DreamZero-DROID) +# TOKENIZER_PATH - umt5-xxl tokenizer (default: /fsx/models/umt5-xxl) +# DATASET_PATH - LeRobot-layout LIBERO dataset (default: /fsx/datasets/libero) +# METADATA_PATH - LIBERO normalization metadata.json +# (default: /fsx/models/metadata-libero.json, generated by +# generate-metadata.yaml). Set empty to fall back to the +# checkpoint's bundled metadata (oxe_droid only). +# CKPT_PATH - Path to write checkpoints (default: /fsx/checkpoints) +# EXPERIMENT_NAME - Experiment / log dir name (default: dreamzero-libero-sft) +# NUM_NODES - Number of nodes in the Ray cluster (default: 2) +# HYDRA_OVERRIDES - Extra Hydra overrides (e.g. "runner.max_steps=1") +# ============================================================================= +set -euo pipefail +set -x + +# --- Configuration (override via environment variables) --- +VENV_NAME="${VENV_NAME:-dreamzero}" +CONFIG_NAME="${CONFIG_NAME:-libero_sft_dreamzero_14b}" + +MODEL_PATH="${MODEL_PATH:-/fsx/models/DreamZero-DROID}" +TOKENIZER_PATH="${TOKENIZER_PATH:-/fsx/models/umt5-xxl}" +DATASET_PATH="${DATASET_PATH:-/fsx/datasets/libero}" +METADATA_PATH="${METADATA_PATH:-/fsx/models/metadata-libero.json}" +CKPT_PATH="${CKPT_PATH:-/fsx/checkpoints}" +EXPERIMENT_NAME="${EXPERIMENT_NAME:-dreamzero-libero-sft}" + +NUM_NODES="${NUM_NODES:-2}" + +# --- Activate the correct venv --- +# Upstream RLinf images use the multi-venv pattern (uv). DreamZero LIBERO SFT +# runs in the dedicated 'dreamzero' venv. Pre-set PYTHONPATH before sourcing the activate +# script so it does not trip the "unbound variable" guard under set -u +# (documented gotcha in examples/AGENTS.md). +export PYTHONPATH="${PYTHONPATH:-}" +UV_PATH="${UV_PATH:-/opt/venv}" + +if [ -f "${UV_PATH}/${VENV_NAME}/bin/activate" ]; then + echo "Activating venv: ${VENV_NAME}" + source "${UV_PATH}/${VENV_NAME}/bin/activate" +elif [ -f "/usr/local/bin/switch_env" ]; then + echo "Activating venv via switch_env: ${VENV_NAME}" + source switch_env "${VENV_NAME}" +else + echo "WARNING: No venv found for ${VENV_NAME}, using system Python" +fi + +echo "Python: $(which python3)" +echo "PyTorch: $(python3 -c 'import torch; print(torch.__version__)' 2>/dev/null || echo 'not found')" + +# --- DreamZero groot package on PYTHONPATH --- +export DREAMZERO_PATH="${DREAMZERO_PATH:-/workspace/DreamZero}" +export PYTHONPATH="${DREAMZERO_PATH}:${PYTHONPATH}" + +# --- EMBODIED_PATH for Hydra searchpath interpolation --- +# Upstream SFT configs interpolate this to resolve the config searchpath. +export EMBODIED_PATH="${EMBODIED_PATH:-/workspace/RLinf/examples/sft}" + +# --- Headless rendering defaults --- +# SFT may not need GL, but osmesa is the safe EKS default and matches the eval +# path (avoids EGL/driver surprises if any sim import touches rendering). +export MUJOCO_GL="${MUJOCO_GL:-osmesa}" +export PYOPENGL_PLATFORM="${PYOPENGL_PLATFORM:-osmesa}" + +# --- Pre-flight checks: verify staged inputs exist --- +for p in "${MODEL_PATH}" "${TOKENIZER_PATH}" "${DATASET_PATH}"; do + if [ ! -e "${p}" ]; then + echo "ERROR: required path not found: ${p}" + echo "Run the model-download / staging job first." + exit 1 + fi +done + +# --- Launch training --- +cd /workspace/RLinf + +LOG_DIR="${CKPT_PATH}/${EXPERIMENT_NAME}" +mkdir -p "${LOG_DIR}" + +# Build Hydra override args (space-separated). These keys match the upstream +# libero_sft_dreamzero_14b.yaml structure. +HYDRA_ARGS="actor.model.model_path=${MODEL_PATH}" +HYDRA_ARGS="${HYDRA_ARGS} actor.model.tokenizer_path=${TOKENIZER_PATH}" +HYDRA_ARGS="${HYDRA_ARGS} data.train_data_paths=${DATASET_PATH}" +HYDRA_ARGS="${HYDRA_ARGS} cluster.num_nodes=${NUM_NODES}" +HYDRA_ARGS="${HYDRA_ARGS} runner.logger.log_path=${LOG_DIR}" + +# NOTE on checkpoint format for eval: +# RLinf's FSDP saver writes a sharded DCP checkpoint (.distcp + .metadata) under +# {ckpt}/actor/dcp_checkpoint/. The LIBERO eval needs a single .pt; convert the +# DCP to .pt offline (CPU) with examples/dreamzero/manifests/convert-checkpoint.yaml. +# We do NOT use +actor.fsdp_config.save_full_model_weights=true: on 2x p5en the +# rank-0 full-state-dict gather for the 16B model is pathologically slow / stalls. +# DCP-only save + offline convert is faster and more reliable. +# IMPORTANT: ensure FSx has ample free space (>=200GB for a 14B DCP checkpoint); +# a full filesystem truncates torch.save mid-write (inline_container.cc +# "unexpected pos") producing corrupt, unreadable shards. + +# Temporal alignment fix: libero_sft_dreamzero_14b.yaml sets action_horizon=16 +# but inherits num_action_per_block=24 from model/dreamzero_14b.yaml (a DROID +# default). For LIBERO the two must match (the working libero_sft_dreamzero_5b +# config sets BOTH to 16). Without this, the forward pass asserts: +# actions.shape[1] / (noise.shape[1]-1) == num_action_per_block // num_frame_per_block +# (got 64/8=8, expected 24//2=12). With num_action_per_block=16: 16//2=8 == 8. +HYDRA_ARGS="${HYDRA_ARGS} actor.model.num_action_per_block=16" + +# Metadata handling: set metadata_json_path when METADATA_PATH is non-empty. +# It now DEFAULTS to /fsx/models/metadata-libero.json (LIBERO libero_sim stats +# generated by generate-metadata.yaml), so the override is passed by default. +# The DreamZero-DROID checkpoint only bundles oxe_droid metadata, so LIBERO SFT +# would fail with a KeyError on embodiment_tag 'libero_sim' without this. Set +# METADATA_PATH="" to fall back to the checkpoint's bundled metadata instead. +# NOTE: metadata_json_path is COMMENTED OUT in the upstream config (not in the +# Hydra struct), so it must be ADDED with the '+' prefix, not overridden. +if [ -n "${METADATA_PATH}" ]; then + HYDRA_ARGS="${HYDRA_ARGS} +actor.model.metadata_json_path=${METADATA_PATH}" +fi + +# Append user-supplied overrides (e.g., HYDRA_OVERRIDES="runner.max_steps=1") +if [ -n "${HYDRA_OVERRIDES:-}" ]; then + echo "Extra Hydra overrides: ${HYDRA_OVERRIDES}" + HYDRA_ARGS="${HYDRA_ARGS} ${HYDRA_OVERRIDES}" +fi + +echo "=== Launching DreamZero LIBERO SFT (Ray / FSDP2) ===" +echo "Config: ${CONFIG_NAME}" +echo "Venv: ${VENV_NAME}" +echo "Model: ${MODEL_PATH}" +echo "Tokenizer: ${TOKENIZER_PATH}" +echo "Dataset: ${DATASET_PATH}" +echo "Nodes: ${NUM_NODES}" +echo "Hydra args: ${HYDRA_ARGS}" + +# --config-path is "config", resolved by Hydra relative to the train script's +# own location (examples/sft/), where the SFT configs are baked in. +# HYDRA_ARGS is intentionally word-split. +# shellcheck disable=SC2086 +python3 examples/sft/train_vla_sft.py \ + --config-path config \ + --config-name "${CONFIG_NAME}" \ + ${HYDRA_ARGS} From 600d32a7e520690fc9e872f00bf34bf30ba236cc Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Wed, 17 Jun 2026 11:58:47 -0500 Subject: [PATCH 02/45] feat(dreamzero): add self-contained two-stage Dockerfile (MIT-0) + docker build helpers Two-stage EFA-overlay Dockerfile ported from the validated RLinf-on-eks image (deduplicated venv via upstream embodied-libero, DCP-save hotfix patch applied at build). Build context helpers (install_extras.sh, run_training_eks.sh, the DCP patch) live under docker/ (NOT build/, which the repo .gitignore excludes). --- 3.test_cases/pytorch/dreamzero/Dockerfile | 281 ++++++++++++++++++ .../docker/scripts/install_extras.sh | 84 ++++++ .../dcp-save-finalize-besteffort.patch | 46 +++ .../docker/scripts/run_training_eks.sh | 176 +++++++++++ 4 files changed, 587 insertions(+) create mode 100644 3.test_cases/pytorch/dreamzero/Dockerfile create mode 100755 3.test_cases/pytorch/dreamzero/docker/scripts/install_extras.sh create mode 100644 3.test_cases/pytorch/dreamzero/docker/scripts/patches/dcp-save-finalize-besteffort.patch create mode 100644 3.test_cases/pytorch/dreamzero/docker/scripts/run_training_eks.sh diff --git a/3.test_cases/pytorch/dreamzero/Dockerfile b/3.test_cases/pytorch/dreamzero/Dockerfile new file mode 100644 index 000000000..18b502a29 --- /dev/null +++ b/3.test_cases/pytorch/dreamzero/Dockerfile @@ -0,0 +1,281 @@ +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 +# ============================================================================= +# RLinf Container Image for Amazon EKS +# +# Strategy: Two-stage build. +# Stage 1: Build upstream RLinf image using their Dockerfile + BUILD_TARGET. +# Stage 2: Layer EFA networking stack on top for multi-node NCCL on EKS. +# +# This preserves upstream's dependency management (uv, multi-venv, install.sh) +# while adding the AWS EFA stack that upstream doesn't include. +# +# The BUILD_TARGET arg selects which upstream image variant to build: +# - embodied-maniskill_libero (ManiSkill + LIBERO, 6 model venvs) +# - embodied-robotwin (RoboTwin, 3 model venvs) +# - embodied-calvin (CALVIN, 2 model venvs) +# - embodied-metaworld (MetaWorld, 2 model venvs) +# - embodied-behavior-openvlaoft (BEHAVIOR, 1 model venv) +# - reason (Agentic RL, vLLM/SGLang) +# - ... (15 targets total, see upstream docker/Dockerfile) +# +# Build: use kubernetes/libero/setup/build-push.sh, which clones RLinf (pinned +# UPSTREAM_REF) and DreamZero (pinned DREAMZERO_REF), builds the upstream +# embodied-libero image as stage 1, then builds this EFA overlay as stage 2 and +# pushes to your ECR. The build context is this test-case dir and must contain +# ./RLinf, ./DreamZero, and ./docker (scripts + patches). +# ============================================================================= + +# ---- Stage 1: Build upstream RLinf image ---- +ARG BUILD_TARGET=embodied-maniskill_libero +ARG UPSTREAM_DOCKERFILE=docker/Dockerfile + +FROM rlinf-upstream-${BUILD_TARGET} AS upstream +# This FROM is a placeholder -- the actual upstream build happens in CodeBuild +# (see buildspec.yml). CodeBuild builds the upstream image first, tags it as +# rlinf-upstream-${BUILD_TARGET}, then this Dockerfile layers EFA on top. +# For local builds, run the upstream build first: +# cd /path/to/RLinf +# docker build --build-arg BUILD_TARGET=embodied-maniskill_libero \ +# -t rlinf-upstream-embodied-maniskill_libero -f docker/Dockerfile . + + +# ---- Stage 2: EFA networking overlay ---- +# Layer EFA, GDRCopy, NCCL, and OpenMPI onto the upstream RLinf image. +# Recipe proven in nccl-tests/Dockerfile on nvidia/cuda base images. + +FROM rlinf-upstream-${BUILD_TARGET} + +ARG GDRCOPY_VERSION=v2.5.1 +ARG EFA_INSTALLER_VERSION=1.47.0 +ARG NCCL_VERSION=v2.21.5-1 + +ENV DEBIAN_FRONTEND=noninteractive + +# -- Clean slate: remove any pre-existing RDMA/NCCL packages -- +# The upstream CUDA base may ship old ibverbs/NCCL that conflict with EFA. +RUN apt-get update -y && \ + apt-get remove -y --allow-change-held-packages \ + ibverbs-utils libibverbs-dev libibverbs1 libmlx5-1 \ + libnccl2 libnccl-dev 2>/dev/null || true && \ + rm -rf /opt/hpcx /usr/local/mpi && \ + rm -f /etc/ld.so.conf.d/hpcx.conf && \ + ldconfig +ENV OPAL_PREFIX= + +# -- System dependencies for EFA/NCCL build -- +# NOTE: The EFA installer requires libevent-pthreads, libhwloc15, udev, +# environment-modules, and tcl which the upstream RLinf base may not include. +# Install them explicitly to avoid "held broken packages" errors. +RUN apt-get install -y --no-install-recommends \ + autoconf automake build-essential cmake curl \ + environment-modules \ + git gcc gdb kmod \ + libevent-pthreads-2.1-7 libhwloc15 \ + libtool pkg-config \ + openssh-client openssh-server \ + tcl udev \ + && rm -rf /var/lib/apt/lists/* + +# Purge cuda-compat to avoid NCCL crashes with incompatible versions +RUN apt-get purge -y cuda-compat-* 2>/dev/null || true + +# -- SSH config for MPI (required for MPIJob multi-node) -- +RUN mkdir -p /var/run/sshd && \ + sed -i 's/[ #]\(.*StrictHostKeyChecking \).*/ \1no/g' /etc/ssh/ssh_config && \ + echo " UserKnownHostsFile /dev/null" >> /etc/ssh/ssh_config && \ + sed -i 's/#\(StrictModes \).*/\1no/g' /etc/ssh/sshd_config + +# -- Library paths (set early so subsequent installs are found) -- +ENV LD_LIBRARY_PATH=/usr/local/cuda/extras/CUPTI/lib64:/opt/amazon/openmpi/lib:/opt/nccl/build/lib:/opt/amazon/efa/lib:/opt/amazon/ofi-nccl/lib:/opt/gdrcopy/lib:/usr/local/lib:${LD_LIBRARY_PATH:-} +ENV PATH=/opt/amazon/openmpi/bin:/opt/amazon/efa/bin:/opt/gdrcopy/bin:${PATH:-/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin} +ENV LIBRARY_PATH=/opt/gdrcopy/lib:${LIBRARY_PATH:-} +ENV CPATH=/opt/gdrcopy/include:${CPATH:-} + +# ============================================================================= +# GDRCopy -- GPU Direct RDMA copy library (userspace) +# +# Required for GPU-initiated DMA transfers over EFA. The kernel module +# (gdrdrv) must be loaded on the host node; we only build userspace here. +# Documented prerequisite: AGENTS.md Discovery #43. +# ============================================================================= +RUN git clone -b ${GDRCOPY_VERSION} https://github.com/NVIDIA/gdrcopy.git /tmp/gdrcopy && \ + cd /tmp/gdrcopy && \ + make prefix=/opt/gdrcopy install && \ + rm -rf /tmp/gdrcopy + +# ============================================================================= +# EFA Installer 1.47.0 +# +# CRITICAL: Must use 1.47.0+ (not 1.34.0/1.38.0 which bundle old libfabric 1.22). +# Installer 1.47.0 bundles: +# - libfabric 2.4.0amzn1.0 (RDMA + DMA-BUF support) +# - aws-ofi-nccl 1.18.0 (installed to /opt/amazon/ofi-nccl) +# - OpenMPI 4 and 5 +# +# No --minimal flag: we need the full install including aws-ofi-nccl. +# The libevent-pthreads and libhwloc15 deps are installed above. +# See AGENTS.md Discoveries #41-#45. +# ============================================================================= +RUN cd /tmp && \ + curl -O https://efa-installer.amazonaws.com/aws-efa-installer-${EFA_INSTALLER_VERSION}.tar.gz && \ + tar -xf aws-efa-installer-${EFA_INSTALLER_VERSION}.tar.gz && \ + cd aws-efa-installer && \ + ./efa_installer.sh -y -g -d --skip-kmod --skip-limit-conf --no-verify && \ + rm -rf /tmp/aws-efa-installer* + +# Verify aws-ofi-nccl was installed by the EFA installer +RUN echo "Verifying AWS OFI NCCL plugin installation..." && \ + (ls -la /opt/amazon/ofi-nccl/lib/x86_64-linux-gnu/libnccl-ofi*.so || \ + ls -la /opt/amazon/ofi-nccl/lib/aarch64-linux-gnu/libnccl-ofi*.so || \ + ls -la /opt/amazon/ofi-nccl/lib/libnccl-ofi*.so) + +# ============================================================================= +# NCCL from source +# +# Build NCCL for target GPU architectures. +# sm_80 (A100), sm_86 (A10G), sm_89 (L4/L40S), sm_90 (H100) +# Note: sm_100 (B200) requires CUDA 12.8+; omit for CUDA 12.4 base. +# ============================================================================= +RUN git clone -b ${NCCL_VERSION} https://github.com/NVIDIA/nccl.git /opt/nccl && \ + cd /opt/nccl && \ + make -j $(nproc) src.build CUDA_HOME=/usr/local/cuda \ + NVCC_GENCODE="-gencode=arch=compute_80,code=sm_80 \ + -gencode=arch=compute_86,code=sm_86 \ + -gencode=arch=compute_89,code=sm_89 \ + -gencode=arch=compute_90,code=sm_90" + +# -- OpenMPI tuning -- +ENV OMPI_MCA_pml=^ucx \ + OMPI_MCA_btl=tcp,self \ + OMPI_MCA_btl_tcp_if_exclude=lo,docker0,veth_def_agent \ + OPAL_PREFIX=/opt/amazon/openmpi \ + NCCL_SOCKET_IFNAME=^docker,lo,veth \ + PMIX_MCA_gds=hash + +# Preload source-built NCCL over any system NCCL +ENV LD_PRELOAD=/opt/nccl/build/lib/libnccl.so + +# -- EFA runtime defaults -- +ENV FI_EFA_USE_HUGE_PAGE=0 +ENV NCCL_TUNER_PLUGIN=/opt/amazon/ofi-nccl/lib/libnccl-ofi-tuner.so + +# -- EKS-specific environment -- +# MuJoCo / ManiSkill headless rendering (osmesa, not egl -- Discovery #8) +ENV MUJOCO_GL=osmesa +ENV PYOPENGL_PLATFORM=osmesa + +# NVIDIA driver capabilities -- required for ManiSkill GPU rendering +ENV NVIDIA_DRIVER_CAPABILITIES=all + +# PyTorch / training defaults +ENV PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True +ENV TOKENIZERS_PARALLELISM=true +ENV TORCH_NCCL_AVOID_RECORD_STREAMS=1 +ENV NCCL_DEBUG=WARN + +# -- Copy EKS-specific scripts -- +COPY docker/scripts/ /workspace/eks/scripts/ + +# ============================================================================= +# Optional extras (controlled by EXTRAS build arg) +# +# EXTRAS is a comma-separated list of packages to install on top of the EFA +# overlay. This allows the same Dockerfile to produce images for different +# examples without rebuilding the expensive EFA/NCCL layers. +# +# Supported values: +# rlinf - RLinf source (required for all training examples) +# +# DreamZero note: the DreamZero `groot` package is provided by the +# COPY DreamZero/ tree on PYTHONPATH (DREAMZERO_PATH). Its `dreamzero` venv is +# built by the upstream embodied-libero target (RLinf PR #1272), so this overlay +# no longer rebuilds it. +# +# Examples: +# --build-arg EXTRAS=rlinf (ManiSkill, LIBERO, pi0, DreamZero) +# ============================================================================= +ARG EXTRAS="rlinf" + +# Copy source repos into build context. +# RLinf is always available (cloned in pre_build). +# DreamZero (groot package) is always cloned by the buildspec (pre_build) and +# made available on PYTHONPATH via DREAMZERO_PATH; it is no longer gated by EXTRAS. +COPY RLinf/ /workspace/RLinf/ +COPY DreamZero/ /workspace/DreamZero/ + +# Apply EKS hotfix patches to the upstream RLinf source. +# dcp-save-finalize-besteffort.patch: tolerate torch DCP's post-write +# finalization broadcast crash (UnpicklingError on multi-node gloo PG) when the +# on-disk checkpoint is already complete. Must be baked into the image because +# Ray worker actors on every node import from /workspace/RLinf. git is present +# (installed in the EFA stage). Patches are validated to apply against the pinned +# UPSTREAM_REF; the build fails loudly if a patch no longer applies. +RUN cd /workspace/RLinf && \ + for p in /workspace/eks/scripts/patches/*.patch; do \ + echo "Applying patch: ${p}"; \ + git apply --verbose "${p}" || patch -p1 < "${p}"; \ + done + +# Install extras based on EXTRAS arg +RUN chmod +x /workspace/eks/scripts/install_extras.sh && \ + /workspace/eks/scripts/install_extras.sh "${EXTRAS}" + +# ============================================================================= +# Security hardening / image hygiene (runs last so it cleans everything above) +# +# 1. Apply available OS security upgrades for fixable HIGH-severity CVEs +# (gnupg/gpg family CVE-2025-68973, linux-libc-dev kernel headers). +# 2. Purge build-time caches (uv/pip/git checkouts under /opt/venv/.cache and +# /root/.cache). These are NOT needed at runtime and account for the bulk of +# third-party CVE/misconfig findings (vendored Go/Rust binaries and cached +# upstream Dockerfiles from transitive dependencies). +# 3. Remove bundled Dockerfiles shipped inside installed dependency source trees +# (gr00t, dexbotic, RLinf, nsight). They are upstream artifacts, never built +# here, and only trigger IaC-misconfig findings. +# +# Trivy evidence: this removes the fixable CRITICAL (gRPC in wandb-core), +# ~36 cache vulnerabilities, and all 55 HIGH Dockerfile misconfigurations. +# ============================================================================= +RUN apt-get update -y && \ + apt-get install -y --only-upgrade --no-install-recommends \ + dirmngr gnupg gnupg-l10n gnupg-utils gnupg2 \ + gpg gpg-agent gpg-wks-client gpg-wks-server gpgconf gpgsm gpgv \ + openssl libssl3 \ + linux-libc-dev 2>/dev/null || true && \ + apt-get clean && \ + rm -rf /var/lib/apt/lists/* && \ + rm -rf /opt/venv/.cache /root/.cache /root/.cargo/registry /tmp/* && \ + find /opt/venv -path '*/.cache' -type d -prune -exec rm -rf {} + 2>/dev/null || true && \ + find /opt /workspace /usr -type f \ + \( -iname 'Dockerfile' -o -iname 'Dockerfile.*' -o -iname '*.Dockerfile' \) \ + -delete 2>/dev/null || true && \ + ldconfig + +# Upgrade build/packaging utilities (and their setuptools-vendored copies) that +# carry fixable HIGH CVEs (jaraco.context CVE-2026-23949, wheel CVE-2026-24049). +# These are used only at package-build time. Run across every model venv. +# A final cache purge runs LAST here: pip/uv re-create /opt/venv/.cache during the +# upgrade above (after the hardening block's purge), and that regenerated cache +# carries third-party files that trip the scanner (e.g. a JWT-shaped string in +# scikit-image's data fetcher). Purging at the very end keeps the image clean. +ENV PIP_NO_CACHE_DIR=1 UV_NO_CACHE=1 +RUN for v in /opt/venv/*/bin/activate; do \ + [ -f "$v" ] || continue; \ + venv_dir="$(dirname "$(dirname "$v")")"; \ + echo "Upgrading packaging tools in ${venv_dir}"; \ + set +u; . "$v"; set -u; \ + pip install --no-cache-dir --upgrade 'setuptools>=78.1.1' 'wheel>=0.46.2' 2>/dev/null || \ + echo " WARNING: tooling upgrade skipped in ${venv_dir}"; \ + deactivate 2>/dev/null || true; \ + done && \ + command -v uv >/dev/null 2>&1 && uv cache clean 2>/dev/null || true; \ + rm -rf /opt/venv/.cache /root/.cache /tmp/* && \ + find /opt/venv -path '*/.cache' -type d -prune -exec rm -rf {} + 2>/dev/null || true + +WORKDIR /workspace/RLinf + +# DreamZero groot package is provided via PYTHONPATH at runtime (see launchers). +ENV DREAMZERO_PATH=/workspace/DreamZero + +CMD ["/bin/bash"] diff --git a/3.test_cases/pytorch/dreamzero/docker/scripts/install_extras.sh b/3.test_cases/pytorch/dreamzero/docker/scripts/install_extras.sh new file mode 100755 index 000000000..c4bc12edb --- /dev/null +++ b/3.test_cases/pytorch/dreamzero/docker/scripts/install_extras.sh @@ -0,0 +1,84 @@ +#!/bin/bash +# ============================================================================= +# Install optional extras for RLinf-on-EKS container image +# +# Called by Dockerfile with EXTRAS arg (comma-separated list). +# Each extra is a self-contained install block. +# +# Supported extras: +# rlinf - RLinf source (editable install in all venvs) +# +# DreamZero deps are no longer installed here -- they come from upstream +# install.sh --env wan (stage 1) plus the DreamZero tree on PYTHONPATH. +# +# Usage (from Dockerfile): +# RUN /tmp/install_extras.sh "rlinf" +# ============================================================================= +set -euo pipefail + +EXTRAS="${1:-rlinf}" +VENVS="/opt/venv/openvla /opt/venv/openvla-oft /opt/venv/openpi" + +echo "=== Installing extras: ${EXTRAS} ===" + +install_in_venvs() { + local pkg_dir=$1 + local pkg_name=$2 + shift 2 + # Remaining args are extra pip packages to install + local extra_pkgs=("$@") + + for venv in ${VENVS}; do + if [ -d "$venv" ]; then + echo " Installing ${pkg_name} into $(basename $venv) venv..." + # Venv activate scripts reference PYTHONPATH/CPATH which may be unset + set +u + # shellcheck disable=SC1091 + . "$venv/bin/activate" + set -u + pip install --no-deps -e "$pkg_dir" 2>/dev/null + for pkg in "${extra_pkgs[@]}"; do + if [ -n "$pkg" ]; then + echo " Extra: $pkg" + MAX_JOBS=4 pip install --no-build-isolation "$pkg" 2>/dev/null || \ + echo " WARNING: $pkg install failed in $(basename $venv), skipping" + fi + done + set +u + deactivate + set -u + fi + done +} + +# Parse comma-separated EXTRAS into array +IFS=',' read -ra EXTRA_LIST <<< "$EXTRAS" + +for extra in "${EXTRA_LIST[@]}"; do + extra=$(echo "$extra" | tr -d ' ') # trim whitespace + case "$extra" in + rlinf) + echo "" + echo "--- Installing: RLinf source ---" + if [ -f /workspace/RLinf/pyproject.toml ]; then + echo " RLinf source found, installing..." + install_in_venvs /workspace/RLinf "RLinf" + else + echo " ERROR: RLinf source not found at /workspace/RLinf." + echo " Ensure buildspec clones RLinf and Dockerfile COPY's it." + exit 1 + fi + ;; + + "") + # Empty string from trailing comma, ignore + ;; + + *) + echo " WARNING: Unknown extra '$extra', skipping." + ;; + esac +done + +echo "" +echo "=== Extras installation complete ===" diff --git a/3.test_cases/pytorch/dreamzero/docker/scripts/patches/dcp-save-finalize-besteffort.patch b/3.test_cases/pytorch/dreamzero/docker/scripts/patches/dcp-save-finalize-besteffort.patch new file mode 100644 index 000000000..6b8c9a744 --- /dev/null +++ b/3.test_cases/pytorch/dreamzero/docker/scripts/patches/dcp-save-finalize-besteffort.patch @@ -0,0 +1,46 @@ +diff --git a/rlinf/hybrid_engines/fsdp/strategy/base.py b/rlinf/hybrid_engines/fsdp/strategy/base.py +index 06bdd2d4..3f757111 100644 +--- a/rlinf/hybrid_engines/fsdp/strategy/base.py ++++ b/rlinf/hybrid_engines/fsdp/strategy/base.py +@@ -233,10 +233,37 @@ class FSDPStrategyBase(ABC): + from torch.distributed import checkpoint as dcp + + dcp_save_path = os.path.join(save_path, "dcp_checkpoint") +- dcp.save( +- {"fsdp_checkpoint": training_state}, +- checkpoint_id=dcp_save_path, +- ) ++ try: ++ dcp.save( ++ {"fsdp_checkpoint": training_state}, ++ checkpoint_id=dcp_save_path, ++ ) ++ except BaseException as dcp_err: ++ # EKS hotfix: torch DCP's post-write finalization broadcast ++ # (all_reduce("write", ...) -> broadcast_object_list) can raise ++ # `_pickle.UnpicklingError: invalid load key, '\x00'` on multi-node ++ # gloo process groups, AFTER write_data() and finish_checkpoint() ++ # have already written every shard + .metadata to disk. Treat the ++ # save as successful when the on-disk DCP is complete; otherwise ++ # re-raise (genuine write failure, e.g. truncated/full filesystem). ++ import glob as _glob ++ ++ meta_ok = os.path.isfile( ++ os.path.join(dcp_save_path, ".metadata") ++ ) ++ shards = _glob.glob(os.path.join(dcp_save_path, "*.distcp")) ++ world = torch.distributed.get_world_size() ++ if meta_ok and len(shards) >= world: ++ if hasattr(cls, "logger") and cls.logger is not None: ++ cls.logger.warning( ++ "dcp.save raised in post-write finalization " ++ f"({type(dcp_err).__name__}: {dcp_err}) but the " ++ f"on-disk checkpoint is complete (.metadata + " ++ f"{len(shards)} shards >= world_size {world}) at " ++ f"{dcp_save_path}; treating save as successful." ++ ) ++ else: ++ raise + + except BaseException as e: + import traceback diff --git a/3.test_cases/pytorch/dreamzero/docker/scripts/run_training_eks.sh b/3.test_cases/pytorch/dreamzero/docker/scripts/run_training_eks.sh new file mode 100644 index 000000000..2df8db654 --- /dev/null +++ b/3.test_cases/pytorch/dreamzero/docker/scripts/run_training_eks.sh @@ -0,0 +1,176 @@ +#!/bin/bash +# ============================================================================= +# RLinf Training Launch Script for Amazon EKS +# +# Generic launcher that works with ANY RLinf embodiment config. +# The upstream multi-venv pattern is preserved: set VENV_NAME to select the +# model's Python environment, then CONFIG_NAME to select the Hydra config. +# +# Expects: +# - Container built from the two-stage Dockerfile (upstream RLinf + EFA) +# - Pre-trained model weights at MODEL_PATH (on shared storage) +# - ManiSkill/simulator assets available (via model-download job or baked in) +# - Shared storage at /fsx (FSx for Lustre PVC) +# - EFA environment variables set via pod spec +# +# Key environment variables (set via K8s manifest env): +# CONFIG_NAME - Hydra config name (e.g., maniskill_ppo_openvla_quickstart) +# VENV_NAME - Python venv to activate (e.g., openvla, openvla-oft, openpi) +# MODEL_PATH - Path to pre-trained model weights on shared storage +# CKPT_PATH - Path to write checkpoints +# NUM_GPUS - GPUs per node +# NUM_NODES - Number of nodes +# ============================================================================= +set -euo pipefail +set -x + +# --- Configuration (override via environment variables) --- +EXPERIMENT_NAME="${EXPERIMENT_NAME:-rlinf-eks}" +CONFIG_NAME="${CONFIG_NAME:-maniskill_ppo_openvla_quickstart}" +VENV_NAME="${VENV_NAME:-openvla}" + +MODEL_PATH="${MODEL_PATH:-/fsx/models/openvla-7b-rlvla-warmup}" +CKPT_PATH="${CKPT_PATH:-/fsx/checkpoints}" + +NUM_GPUS="${NUM_GPUS:-8}" +NUM_NODES="${NUM_NODES:-1}" + +# Component placement string (e.g., "0-7" for 8 GPUs) +GPU_RANGE="0-$((NUM_GPUS - 1))" +COMPONENT_PLACEMENT="${COMPONENT_PLACEMENT:-actor,env,rollout: ${GPU_RANGE}}" + +# --- Activate the correct venv --- +# Upstream RLinf images use multi-venv pattern with switch_env utility. +# Each model (openvla, openvla-oft, openpi, gr00t, etc.) has its own venv. +UV_PATH="${UV_PATH:-/opt/venv}" + +# Pre-set variables that venv activate scripts may reference but are not +# guaranteed to exist in container environments (avoids "unbound variable" +# errors under set -u). +export PYTHONPATH="${PYTHONPATH:-}" + +if [ -f "${UV_PATH}/${VENV_NAME}/bin/activate" ]; then + echo "Activating venv: ${VENV_NAME}" + source "${UV_PATH}/${VENV_NAME}/bin/activate" +elif [ -f "/usr/local/bin/switch_env" ]; then + echo "Activating venv via switch_env: ${VENV_NAME}" + source switch_env "${VENV_NAME}" +else + echo "WARNING: No venv found for ${VENV_NAME}, using system Python" +fi + +echo "Python: $(which python3)" +echo "PyTorch: $(python3 -c 'import torch; print(torch.__version__)' 2>/dev/null || echo 'not found')" + +# --- Environment (defaults set in Dockerfile, override via pod spec) --- +export NCCL_DEBUG="${NCCL_DEBUG:-WARN}" +export PYTORCH_CUDA_ALLOC_CONF="${PYTORCH_CUDA_ALLOC_CONF:-expandable_segments:True}" +export TOKENIZERS_PARALLELISM="${TOKENIZERS_PARALLELISM:-true}" + +# Ensure libcuda.so is discoverable by subprocesses (Ray env offload workers). +# The NVIDIA runtime mounts it at /usr/lib64 but that may not be in LD_LIBRARY_PATH. +export LD_LIBRARY_PATH="/usr/lib64:${LD_LIBRARY_PATH:-}" + +# ManiSkill / MuJoCo headless rendering +export MUJOCO_GL="${MUJOCO_GL:-osmesa}" +export PYOPENGL_PLATFORM="${PYOPENGL_PLATFORM:-osmesa}" +export NVIDIA_DRIVER_CAPABILITIES="${NVIDIA_DRIVER_CAPABILITIES:-all}" + +# EFA (set in pod spec, but provide defaults) +export FI_PROVIDER="${FI_PROVIDER:-efa}" +export FI_EFA_USE_DEVICE_RDMA="${FI_EFA_USE_DEVICE_RDMA:-1}" +export FI_EFA_FORK_SAFE="${FI_EFA_FORK_SAFE:-1}" + +# --- Link simulator assets if available --- +# The upstream image provides link_assets for ManiSkill/SAPIEN symlinks +if [ -x "/usr/local/bin/link_assets" ]; then + link_assets +fi + +# --- Pre-step: Verify model weights --- +if [ -n "${MODEL_PATH}" ] && [ "${MODEL_PATH}" != "none" ]; then + if [ ! -f "$MODEL_PATH/config.json" ] && [ ! -f "$MODEL_PATH/model.safetensors" ]; then + echo "ERROR: Model weights not found at $MODEL_PATH" + echo "Expected config.json or model.safetensors. Run the model-download job first." + exit 1 + fi +fi + +# Clear stale Python bytecode from previous runs (FSx is persistent storage) +find "$CKPT_PATH" -name "__pycache__" -exec rm -rf {} + 2>/dev/null || true + +# Clear HuggingFace transformers dynamic module cache +rm -rf /root/.cache/huggingface/modules/transformers_modules/ 2>/dev/null || true + +# --- Launch training --- +echo "=== Launching RLinf training ===" +echo "Config: ${CONFIG_NAME}" +echo "Venv: ${VENV_NAME}" +echo "Model: ${MODEL_PATH}" +echo "GPUs: ${NUM_GPUS} x ${NUM_NODES} nodes" + +cd /workspace/RLinf + +# Set EMBODIED_PATH for Hydra config interpolation (used in upstream configs +# to resolve relative paths to examples/embodiment/). +export EMBODIED_PATH="${EMBODIED_PATH:-/workspace/RLinf/examples/embodiment}" + +# Build Hydra override args. +# Start with model paths and cluster config, then append any user-supplied overrides. +# HYDRA_OVERRIDES env var allows callers (e.g., validation harness) to inject +# additional overrides like "runner.max_steps=1" without modifying this script. +HYDRA_ARGS="" +if [ -n "${MODEL_PATH}" ] && [ "${MODEL_PATH}" != "none" ]; then + HYDRA_ARGS="actor.model.model_path=${MODEL_PATH} rollout.model.model_path=${MODEL_PATH}" +fi +HYDRA_ARGS="${HYDRA_ARGS} cluster.num_nodes=${NUM_NODES}" + +# Append user-supplied overrides (e.g., HYDRA_OVERRIDES="runner.max_steps=1") +if [ -n "${HYDRA_OVERRIDES:-}" ]; then + echo "Extra Hydra overrides: ${HYDRA_OVERRIDES}" + HYDRA_ARGS="${HYDRA_ARGS} ${HYDRA_OVERRIDES}" +fi + +# Support YAML config override file for complex keys (e.g., component_placement +# with commas in the key name that Hydra CLI cannot parse). +# Set HYDRA_CONFIG_FILE to a YAML file path; its contents will be patched into +# the base config file before launching training. The container is ephemeral, +# so in-place modification is safe. +if [ -n "${HYDRA_CONFIG_FILE:-}" ] && [ -f "${HYDRA_CONFIG_FILE}" ]; then + CONFIG_FILE="examples/embodiment/config/${CONFIG_NAME}.yaml" + echo "Patching config with override file: ${HYDRA_CONFIG_FILE}" + python3 -c " +import yaml, sys +with open('${CONFIG_FILE}') as f: + base = yaml.safe_load(f) +with open('${HYDRA_CONFIG_FILE}') as f: + override = yaml.safe_load(f) + +def deep_merge(base, override): + for k, v in override.items(): + if k in base and isinstance(base[k], dict) and isinstance(v, dict): + deep_merge(base[k], v) + else: + base[k] = v + +deep_merge(base, override) +with open('${CONFIG_FILE}', 'w') as f: + yaml.dump(base, f, default_flow_style=False, sort_keys=False) +print(f'Patched {len(override)} top-level keys into ${CONFIG_FILE}') +" +fi + +LOG_DIR="${CKPT_PATH}/${EXPERIMENT_NAME}" +HYDRA_ARGS="${HYDRA_ARGS} runner.logger.log_path=${LOG_DIR}" + +echo "Hydra args: ${HYDRA_ARGS}" + +# Call train_embodied_agent.py directly (not run_embodiment.sh) so we can pass +# arbitrary Hydra overrides. run_embodiment.sh does not forward extra args. +# --config-path is relative to the script's directory (examples/embodiment/), +# so use just "config" not the full path from repo root. +# shellcheck disable=SC2086 +python3 examples/embodiment/train_embodied_agent.py \ + --config-path config \ + --config-name "${CONFIG_NAME}" \ + ${HYDRA_ARGS} From 55ede04777041312c652f59d582724f2a294ce22 Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Wed, 17 Jun 2026 12:02:57 -0500 Subject: [PATCH 03/45] feat(dreamzero): add workflow manifests (RayJob SFT, download, metadata, convert, eval) --- .../kubernetes/libero/convert-checkpoint.yaml | 78 ++++++ .../kubernetes/libero/dreamzero-eval.yaml | 154 ++++++++++++ .../kubernetes/libero/dreamzero-sft.yaml | 225 ++++++++++++++++++ .../kubernetes/libero/generate-metadata.yaml | 93 ++++++++ .../kubernetes/libero/model-download.yaml | 161 +++++++++++++ 5 files changed, 711 insertions(+) create mode 100644 3.test_cases/pytorch/dreamzero/kubernetes/libero/convert-checkpoint.yaml create mode 100644 3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-eval.yaml create mode 100644 3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-sft.yaml create mode 100644 3.test_cases/pytorch/dreamzero/kubernetes/libero/generate-metadata.yaml create mode 100644 3.test_cases/pytorch/dreamzero/kubernetes/libero/model-download.yaml diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/convert-checkpoint.yaml b/3.test_cases/pytorch/dreamzero/kubernetes/libero/convert-checkpoint.yaml new file mode 100644 index 000000000..6c066f956 --- /dev/null +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/convert-checkpoint.yaml @@ -0,0 +1,78 @@ +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 +# DreamZero Checkpoint Conversion (FSDP DCP shards -> single .pt) +# +# RLinf SFT writes sharded FSDP2 checkpoints (___0.distcp) under +# /fsx/checkpoints///checkpoints/global_step_/actor/dcp_checkpoint/ +# The LIBERO simulator eval (dreamzero-eval) needs a single consolidated .pt +# (runner.ckpt_path). This Job runs upstream's convert_dcp_to_pt.py to produce +# .../global_step_/actor/model_state_dict/full_weights.pt +# +# CPU-only (no GPU). Single node. +# +# Prerequisites: +# - SFT run produced a global_step_ checkpoint on FSx +# - Training image built/pushed +# +# Usage (restricted envsubst protects the inline ${...} shell vars): +# export ECR_URI=.dkr.ecr..amazonaws.com/ +# export NAMESPACE=natharno +# # Optionally override STEP (default global_step_1): +# kubectl -n $NAMESPACE create configmap dreamzero-convert-launcher \ +# --from-file=convert_checkpoint.sh=kubernetes/libero/scripts/convert_checkpoint.sh \ +# --dry-run=client -o yaml | kubectl apply -f - +# envsubst '${ECR_URI} ${NAMESPACE}' < kubernetes/libero/convert-checkpoint.yaml | kubectl apply -f - +--- +apiVersion: batch/v1 +kind: Job +metadata: + name: dreamzero-convert + namespace: ${NAMESPACE} + labels: + app: dreamzero-convert + example: dreamzero +spec: + backoffLimit: 1 + activeDeadlineSeconds: 3600 + template: + metadata: + labels: + app: dreamzero-convert + spec: + restartPolicy: Never + containers: + - name: convert + image: ${ECR_URI}:latest + command: ["bash", "/opt/scripts/convert_checkpoint.sh"] + env: + - name: VENV_NAME + value: "dreamzero" + - name: CKPT_PATH + value: "/fsx/checkpoints" + - name: EXPERIMENT_NAME + value: "dreamzero-libero-sft" + - name: INNER_EXPERIMENT + value: "libero_sft_dreamzero" + - name: STEP + value: "global_step_1" + resources: + requests: + cpu: "8" + memory: "64Gi" + limits: + cpu: "16" + memory: "128Gi" + volumeMounts: + - name: fsx + mountPath: /fsx + - name: scripts + mountPath: /opt/scripts + readOnly: true + volumes: + - name: fsx + persistentVolumeClaim: + claimName: fsx-claim + - name: scripts + configMap: + name: dreamzero-convert-launcher + optional: false diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-eval.yaml b/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-eval.yaml new file mode 100644 index 000000000..573a754b1 --- /dev/null +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-eval.yaml @@ -0,0 +1,154 @@ +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 +# DreamZero LIBERO 14B Simulator Eval (single-node GPU Job) +# ============================================================================= +# Evaluates a trained DreamZero 14B checkpoint (full_weights.pt) in the LIBERO +# Spatial simulator via upstream examples/embodiment/eval_embodied_agent.py, +# reporting success_once and saving in-sim rollout videos. +# +# This is a single-pod GPU Job (NOT a StatefulSet): the eval config pins +# cluster.num_nodes=1, so there is no Ray head/worker split. The 14B model is +# sharded across the 8 GPUs of one p5en.48xlarge node with single-node FSDP. +# +# Prerequisites: +# - SFT checkpoint converted to a single .pt (convert-checkpoint.yaml), at +# /fsx/checkpoints/dreamzero-libero-sft/libero_sft_dreamzero/checkpoints/ +# global_step_1/actor/model_state_dict/full_weights.pt +# - LIBERO normalization metadata generated (generate-metadata.yaml) at +# /fsx/models/metadata-libero.json +# - umt5-xxl tokenizer + DreamZero-DROID backbone staged on FSx +# (/fsx/models/umt5-xxl, /fsx/models/DreamZero-DROID) +# - LIBERO dataset/assets staged (model-download.yaml) +# - Container image built and pushed to ECR (examples/buildspec.yml) +# - FSx PVC "fsx-claim" bound; training-sa ServiceAccount exists +# +# ConfigMaps (create BOTH before applying this manifest): +# # 1) the eval launcher script: +# kubectl -n ${NAMESPACE} create configmap dreamzero-eval-launcher \ +# --from-file=run_dreamzero_eval_eks.sh=kubernetes/libero/scripts/run_dreamzero_eval_eks.sh \ +# --dry-run=client -o yaml | kubectl apply -f - +# # 2) the 14B eval config (mounted at /opt/eval-config, copied into the +# # embodiment config dir by the launcher so upstream config groups stay visible): +# kubectl -n ${NAMESPACE} create configmap dreamzero-eval-config \ +# --from-file=libero_spatial_eval_dreamzero_14b.yaml=kubernetes/libero/scripts/libero_spatial_eval_dreamzero_14b.yaml \ +# --dry-run=client -o yaml | kubectl apply -f - +# +# Apply (RESTRICTED envsubst -- substitute ONLY ${ECR_URI} and ${NAMESPACE} so the +# inline ${PYTHONPATH:-} guards in the bootstrap are NOT clobbered): +# export ECR_URI=.dkr.ecr..amazonaws.com/ +# export NAMESPACE=natharno +# envsubst '${ECR_URI} ${NAMESPACE}' < kubernetes/libero/dreamzero-eval.yaml | kubectl apply -f - +--- +apiVersion: batch/v1 +kind: Job +metadata: + name: dreamzero-eval + namespace: ${NAMESPACE} + labels: + app: dreamzero-eval + reference: rlinf + example: dreamzero-eval +spec: + backoffLimit: 0 + ttlSecondsAfterFinished: 86400 + template: + metadata: + labels: + app: dreamzero-eval + example: dreamzero-eval + annotations: + karpenter.sh/do-not-disrupt: "true" + spec: + # Job: do not restart-loop on eval failure. + restartPolicy: Never + serviceAccountName: training-sa + tolerations: + - key: nvidia.com/gpu + operator: Exists + effect: NoSchedule + terminationGracePeriodSeconds: 120 + containers: + - name: eval + image: ${ECR_URI}:latest + command: + - bash + - /opt/scripts/run_dreamzero_eval_eks.sh + env: + # --- DreamZero LIBERO 14B eval config --- + - name: VENV_NAME + value: "dreamzero" + - name: CONFIG_NAME + value: "libero_spatial_eval_dreamzero_14b" + # full_weights.pt produced by convert-checkpoint.yaml from the SFT DCP. + - name: CKPT_PT + value: "/fsx/checkpoints/dreamzero-libero-sft/libero_sft_dreamzero/checkpoints/global_step_1/actor/model_state_dict/full_weights.pt" + # LIBERO (libero_sim) normalization metadata, generated by generate-metadata.yaml. + - name: METADATA_PATH + value: "/fsx/models/metadata-libero.json" + - name: TOKENIZER_PATH + value: "/fsx/models/umt5-xxl" + - name: MODEL_PATH + value: "/fsx/models/DreamZero-DROID" + - name: LOG_DIR + value: "/fsx/checkpoints/dreamzero-libero-eval" + - name: SAVE_VIDEO + value: "True" + - name: POD_NAMESPACE + valueFrom: + fieldRef: + fieldPath: metadata.namespace + # --- NCCL + EFA (matches dreamzero-sft.yaml) --- + - name: NCCL_DEBUG + value: "INFO" + - name: FI_PROVIDER + value: "efa" + - name: FI_EFA_USE_DEVICE_RDMA + value: "1" + - name: FI_EFA_FORK_SAFE + value: "1" + - name: PYTORCH_CUDA_ALLOC_CONF + value: "expandable_segments:True" + - name: TOKENIZERS_PARALLELISM + value: "true" + - name: TORCH_NCCL_AVOID_RECORD_STREAMS + value: "1" + - name: NVIDIA_DRIVER_CAPABILITIES + value: "all" + resources: + requests: + nvidia.com/gpu: 8 + vpc.amazonaws.com/efa: 16 + cpu: "90" + memory: "1000Gi" + limits: + nvidia.com/gpu: 8 + vpc.amazonaws.com/efa: 16 + cpu: "96" + memory: "1100Gi" + volumeMounts: + - name: fsx + mountPath: /fsx + - name: dshm + mountPath: /dev/shm + - name: scripts + mountPath: /opt/scripts + readOnly: true + - name: eval-config + mountPath: /opt/eval-config + readOnly: true + volumes: + - name: fsx + persistentVolumeClaim: + claimName: fsx-claim + - name: dshm + emptyDir: + medium: Memory + sizeLimit: "128Gi" + - name: scripts + configMap: + name: dreamzero-eval-launcher + optional: false + - name: eval-config + configMap: + name: dreamzero-eval-config + optional: false diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-sft.yaml b/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-sft.yaml new file mode 100644 index 000000000..1e58484f9 --- /dev/null +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-sft.yaml @@ -0,0 +1,225 @@ +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 +# DreamZero LIBERO Multi-Node SFT (KubeRay RayJob, FSDP2) +# +# Supervised fine-tuning of DreamZero (14B WAM) on LIBERO across 2x +# p5en.48xlarge nodes (8x H200 each) using upstream RLinf's Cluster (Ray) +# scheduler with FSDP2 -- warm-started from the DreamZero-DROID checkpoint. +# +# Mechanism: +# - The KubeRay operator starts the Ray HEAD (headGroupSpec) and one WORKER +# (workerGroupSpecs, replicas:1). When the Ray cluster is ready, the operator +# runs `entrypoint` on the head: the Ray-agnostic launcher +# run_dreamzero_sft_eks.sh. RLinf's Cluster scheduler fans the FSDP2 training +# across the whole Ray cluster (no torchrun / DeepSpeed, no manual head +# election). +# - shutdownAfterJobFinishes tears the RayCluster down when the job completes. +# +# Prerequisites: +# - KubeRay operator installed (infrastructure/addons, enable_kuberay=true) +# - dataset/model staging job completed +# (DreamZero-DROID warm-start + umt5-xxl tokenizer + LIBERO dataset on FSx) +# - FSx PVC "fsx-claim" bound, >=250GB free +# - training-sa ServiceAccount exists +# - ConfigMap "dreamzero-sft-launcher" created from the launcher script +# +# Usage: +# kubectl -n ${NAMESPACE} create configmap dreamzero-sft-launcher \ +# --from-file=run_dreamzero_sft_eks.sh=kubernetes/libero/scripts/run_dreamzero_sft_eks.sh \ +# --dry-run=client -o yaml | kubectl apply -f - +# export ECR_URI=.dkr.ecr..amazonaws.com/ +# export NAMESPACE=rlinf +# # RESTRICTED envsubst -- substitute ONLY ${ECR_URI} and ${NAMESPACE}. +# # (The Ray head-election bash is gone, but keep this habit: other inline +# # shell snippets, e.g. the launcher copy, must not be expanded.) +# envsubst '${ECR_URI} ${NAMESPACE}' < kubernetes/libero/dreamzero-sft.yaml | kubectl apply -f - +--- +apiVersion: ray.io/v1 +kind: RayJob +metadata: + name: dreamzero-sft + namespace: ${NAMESPACE} + labels: + app: dreamzero-sft + reference: rlinf + example: dreamzero-sft +spec: + # Run the launcher on the head once the Ray cluster is ready. The launcher + # copies itself from the mounted ConfigMap (if present) then execs training. + entrypoint: >- + bash -c ' + if [ -f /tmp/scripts/run_dreamzero_sft_eks.sh ]; then + cp /tmp/scripts/run_dreamzero_sft_eks.sh /workspace/eks/scripts/run_dreamzero_sft_eks.sh; + chmod +x /workspace/eks/scripts/run_dreamzero_sft_eks.sh; + fi; + bash /workspace/eks/scripts/run_dreamzero_sft_eks.sh' + shutdownAfterJobFinishes: true + ttlSecondsAfterFinished: 600 + # The KubeRay submitter pod runs `ray job submit` against the head. It does NOT + # inherit the head/worker container env, and `ray` lives in the dreamzero venv, + # so without this PATH it fails with "ray: command not found". Reuse the head image. + submitterPodTemplate: + spec: + restartPolicy: Never + containers: + - name: ray-job-submitter + image: ${ECR_URI}:latest + env: + - name: PATH + value: "/opt/venv/dreamzero/bin:/opt/amazon/openmpi/bin:/opt/amazon/efa/bin:/opt/gdrcopy/bin:/usr/local/nvidia/bin:/usr/local/cuda/bin:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin" + rayClusterSpec: + rayVersion: "2.55.1" + headGroupSpec: + rayStartParams: + dashboard-host: "0.0.0.0" + num-gpus: "8" + template: + metadata: + labels: + app: dreamzero-sft + example: dreamzero-sft + annotations: + karpenter.sh/do-not-disrupt: "true" + spec: + serviceAccountName: training-sa + terminationGracePeriodSeconds: 120 + tolerations: + - key: nvidia.com/gpu + operator: Exists + effect: NoSchedule + affinity: + podAntiAffinity: + requiredDuringSchedulingIgnoredDuringExecution: + - labelSelector: + matchExpressions: + - key: app + operator: In + values: + - dreamzero-sft + topologyKey: kubernetes.io/hostname + topologySpreadConstraints: + - maxSkew: 1 + topologyKey: topology.k8s.aws/network-node-layer-2 + whenUnsatisfiable: ScheduleAnyway + labelSelector: + matchLabels: + app: dreamzero-sft + containers: + - name: ray-head + image: ${ECR_URI}:latest + env: &dreamzero_env + # KubeRay runs `ray start` as the container command (non-interactive, + # ~/.bashrc not sourced). `ray` lives in the dreamzero venv, so prepend + # it to PATH or the head/worker crash with "ray: command not found". + - name: PATH + value: "/opt/venv/dreamzero/bin:/opt/amazon/openmpi/bin:/opt/amazon/efa/bin:/opt/gdrcopy/bin:/usr/local/nvidia/bin:/usr/local/cuda/bin:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin" + - name: VENV_NAME + value: "dreamzero" + - name: CONFIG_NAME + value: "libero_sft_dreamzero_14b" + - name: MODEL_PATH + value: "/fsx/models/DreamZero-DROID" + - name: TOKENIZER_PATH + value: "/fsx/models/umt5-xxl" + - name: DATASET_PATH + value: "/fsx/datasets/libero" + - name: METADATA_PATH + value: "/fsx/models/metadata-libero.json" + - name: CKPT_PATH + value: "/fsx/checkpoints" + - name: EXPERIMENT_NAME + value: "dreamzero-libero-sft" + - name: NUM_NODES + value: "2" + - name: NCCL_DEBUG + value: "INFO" + - name: FI_PROVIDER + value: "efa" + - name: FI_EFA_USE_DEVICE_RDMA + value: "1" + - name: FI_EFA_FORK_SAFE + value: "1" + - name: PYTORCH_CUDA_ALLOC_CONF + value: "expandable_segments:True" + - name: TOKENIZERS_PARALLELISM + value: "true" + - name: TORCH_NCCL_AVOID_RECORD_STREAMS + value: "1" + - name: NVIDIA_DRIVER_CAPABILITIES + value: "all" + resources: &dreamzero_resources + requests: + nvidia.com/gpu: 8 + vpc.amazonaws.com/efa: 16 + cpu: "90" + memory: "1000Gi" + limits: + nvidia.com/gpu: 8 + vpc.amazonaws.com/efa: 16 + cpu: "96" + memory: "1100Gi" + volumeMounts: &dreamzero_mounts + - name: fsx + mountPath: /fsx + - name: dshm + mountPath: /dev/shm + - name: launcher + mountPath: /tmp/scripts + readOnly: true + volumes: &dreamzero_volumes + - name: fsx + persistentVolumeClaim: + claimName: fsx-claim + - name: dshm + emptyDir: + medium: Memory + sizeLimit: "128Gi" + - name: launcher + configMap: + name: dreamzero-sft-launcher + optional: false + workerGroupSpecs: + - groupName: worker + replicas: 1 + minReplicas: 1 + maxReplicas: 1 + rayStartParams: + num-gpus: "8" + template: + metadata: + labels: + app: dreamzero-sft + example: dreamzero-sft + annotations: + karpenter.sh/do-not-disrupt: "true" + spec: + serviceAccountName: training-sa + terminationGracePeriodSeconds: 120 + tolerations: + - key: nvidia.com/gpu + operator: Exists + effect: NoSchedule + affinity: + podAntiAffinity: + requiredDuringSchedulingIgnoredDuringExecution: + - labelSelector: + matchExpressions: + - key: app + operator: In + values: + - dreamzero-sft + topologyKey: kubernetes.io/hostname + topologySpreadConstraints: + - maxSkew: 1 + topologyKey: topology.k8s.aws/network-node-layer-2 + whenUnsatisfiable: ScheduleAnyway + labelSelector: + matchLabels: + app: dreamzero-sft + containers: + - name: ray-worker + image: ${ECR_URI}:latest + env: *dreamzero_env + resources: *dreamzero_resources + volumeMounts: *dreamzero_mounts + volumes: *dreamzero_volumes diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/generate-metadata.yaml b/3.test_cases/pytorch/dreamzero/kubernetes/libero/generate-metadata.yaml new file mode 100644 index 000000000..6426eb3a9 --- /dev/null +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/generate-metadata.yaml @@ -0,0 +1,93 @@ +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 +# Generate LIBERO Normalization Metadata for DreamZero SFT +# +# Produces /fsx/models/metadata-libero.json with normalization statistics for +# embodiment `libero_sim`, used by the DreamZero LIBERO 14B SFT job. +# +# WHY THIS IS NEEDED: +# DreamZero LIBERO SFT warm-starts from the DreamZero-DROID checkpoint, which +# bundles experiment_cfg/metadata.json for embodiment `oxe_droid` ONLY. Training +# with embodiment_tag `libero_sim` fails with: +# KeyError: "embodiment_tag 'libero_sim' not found in +# /fsx/models/DreamZero-DROID/experiment_cfg/metadata.json (keys: ['oxe_droid'])." +# This Job generates LIBERO-specific stats via upstream's toolkit so the SFT +# launcher can point actor.model.metadata_json_path at them. +# +# WHY THE TRAINING IMAGE (not python:3.11-slim): +# Metadata generation requires the RLinf repo's toolkit +# (toolkits/lerobot/generate_dreamzero_metadata.py) PLUS the dreamzero venv +# (groot, lerobot, etc.). The slim staging image used by model-download.yaml +# does NOT have these, so this step runs in ${ECR_URI}:latest instead. +# +# OUTPUT: +# Writes /fsx/models/metadata-libero.json (top-level key `libero_sim`). The SFT +# launcher (run_dreamzero_sft_eks.sh) and manifest (dreamzero-sft.yaml) default +# METADATA_PATH to this path, so SFT uses it automatically. +# +# Prerequisites: +# - LIBERO dataset staged on FSx at /fsx/datasets/libero (by model-download.yaml) +# - Training image built and pushed to ECR (${ECR_URI}:latest) +# - FSx PVC "fsx-claim" bound in the target namespace +# +# Usage: +# export ECR_URI=.dkr.ecr..amazonaws.com/ +# export NAMESPACE=natharno +# # NOTE: use RESTRICTED envsubst -- substitute ONLY ${ECR_URI} and ${NAMESPACE}. +# # The container command contains shell vars (${PYTHONPATH}, ${DREAMZERO_PATH}) +# # that unrestricted envsubst would clobber to empty strings. Same pattern as +# # dreamzero-sft.yaml. +# envsubst '${ECR_URI} ${NAMESPACE}' < kubernetes/libero/generate-metadata.yaml | kubectl apply -f - +# kubectl logs -f job/generate-metadata-dreamzero -n $NAMESPACE +--- +apiVersion: batch/v1 +kind: Job +metadata: + name: generate-metadata-dreamzero + namespace: ${NAMESPACE} + labels: + app: generate-metadata + reference: rlinf + example: dreamzero +spec: + backoffLimit: 1 + activeDeadlineSeconds: 3600 + template: + metadata: + labels: + app: generate-metadata + example: dreamzero + spec: + restartPolicy: Never + containers: + - name: generate-metadata + image: ${ECR_URI}:latest + command: + - bash + - -lc + - | + set -euo pipefail + export PYTHONPATH="${PYTHONPATH:-}" + source /opt/venv/dreamzero/bin/activate + export DREAMZERO_PATH=/workspace/DreamZero + export PYTHONPATH="${DREAMZERO_PATH}:${PYTHONPATH}" + cd /workspace/RLinf + python3 toolkits/lerobot/generate_dreamzero_metadata.py \ + --preset libero_sim \ + --dataset-root /fsx/datasets/libero \ + --output-metadata /fsx/models/metadata-libero.json + ls -la /fsx/models/metadata-libero.json + resources: + requests: + cpu: "4" + memory: "16Gi" + limits: + cpu: "8" + memory: "32Gi" + volumeMounts: + - name: fsx + mountPath: /fsx + volumes: + - name: fsx + persistentVolumeClaim: + claimName: fsx-claim diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/model-download.yaml b/3.test_cases/pytorch/dreamzero/kubernetes/libero/model-download.yaml new file mode 100644 index 000000000..dcd798e38 --- /dev/null +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/model-download.yaml @@ -0,0 +1,161 @@ +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 +# Download Model Weights and Dataset for DreamZero LIBERO 14B SFT +# +# Stages the three artifacts needed to fine-tune the DreamZero 14B World-Action +# Model on LIBERO (warm-started from the DreamZero-DROID checkpoint): +# 1. GEAR-Dreams/DreamZero-DROID -> /fsx/models/DreamZero-DROID +# 14B warm-start checkpoint (the DROID-pretrained model we SFT from). +# 2. google/umt5-xxl -> /fsx/models/umt5-xxl +# Text tokenizer / encoder used by the model. +# 3. physical-intelligence/libero (HF DATASET) -> /fsx/datasets/libero +# LIBERO trajectories in LeRobot layout. +# +# GOTCHA: the dataset MUST be `physical-intelligence/libero`, NOT `lerobot/libero`. +# The two repos share a name but have DIFFERENT observation.state / action column +# schemas. Only `physical-intelligence/libero` matches the libero_sim preset's +# state/actions schema. Using `lerobot/libero` will silently train on the wrong +# column layout. +# +# The two models and the LIBERO dataset are downloaded with snapshot_download +# (matching the house pattern used by the sibling example model-download jobs). +# hf_transfer is enabled only for the large model checkpoints (multi-GB files); +# it is intentionally NOT used for the dataset, which is a LeRobot repo containing +# many small files where hf_transfer offers no benefit. +# +# NOTE: LIBERO normalization metadata (embodiment libero_sim) is NOT generated +# here. The DreamZero-DROID checkpoint bundles experiment_cfg/metadata.json for +# embodiment oxe_droid ONLY, so LIBERO SFT needs its own libero_sim stats. That +# metadata is produced by a SEPARATE step -- see generate-metadata.yaml -- which +# runs in the TRAINING image because it requires the RLinf toolkit + the +# dreamzero venv (groot, lerobot, etc.) that this python:3.11-slim staging image +# does NOT have. Run generate-metadata.yaml AFTER this job stages the dataset. +# +# Prerequisites: +# - FSx for Lustre PVC (fsx-claim) available in the target namespace +# - Storage: ~117GB models (umt5-xxl ~52GB + DreamZero-DROID ~65GB) + ~35GB +# LIBERO dataset (~152GB total). Ensure FSx has headroom. +# +# Usage: +# export NAMESPACE=natharno # shared test cluster namespace +# envsubst < kubernetes/libero/model-download.yaml | kubectl apply -f - +# kubectl logs -f job/model-download-dreamzero -n $NAMESPACE +apiVersion: batch/v1 +kind: Job +metadata: + name: model-download-dreamzero + namespace: ${NAMESPACE} + labels: + app: model-download + example: dreamzero +spec: + backoffLimit: 1 + activeDeadlineSeconds: 7200 + template: + spec: + restartPolicy: OnFailure + containers: + - name: downloader + image: python:3.11-slim + command: + - bash + - -c + - | + set -euo pipefail + + echo "============================================" + echo " DreamZero LIBERO 14B SFT: Model & Dataset Download" + echo "============================================" + + pip install -q -U huggingface_hub hf_transfer + + # hf_transfer accelerates LARGE multi-GB file downloads (model + # checkpoints). It is NOT used for the dataset (many small files). + export HF_HUB_ENABLE_HF_TRANSFER=1 + + download_model() { + local model_id=$1 + local target_dir=$2 + echo "" + echo "=== Model: $model_id ===" + echo " Target: $target_dir" + if [ -f "$target_dir/.download_complete" ]; then + echo " Completion sentinel found. Skipping." + return + fi + echo " Downloading..." + python3 -c " + import os + from huggingface_hub import snapshot_download + snapshot_download( + repo_id='$model_id', + local_dir='$target_dir', + token=os.environ.get('HF_TOKEN') or None, + ) + print(' Download complete') + " + # set -e guarantees we only reach this line on a successful download. + touch "$target_dir/.download_complete" + du -sh "$target_dir" | awk '{print " Size: " $1}' + } + + download_dataset() { + local repo_id=$1 + local target_dir=$2 + echo "" + echo "=== Dataset: $repo_id ===" + echo " Target: $target_dir" + if [ -f "$target_dir/.download_complete" ]; then + echo " Completion sentinel found. Skipping." + return + fi + # LeRobot dataset: many small files, so DISABLE hf_transfer here. + HF_HUB_ENABLE_HF_TRANSFER=0 python3 -c " + import os + from huggingface_hub import snapshot_download + snapshot_download( + repo_id='$repo_id', + repo_type='dataset', + local_dir='$target_dir', + token=os.environ.get('HF_TOKEN') or None, + ) + print(' Download complete') + " + # set -e guarantees we only reach this line on a successful download. + touch "$target_dir/.download_complete" + du -sh "$target_dir" | awk '{print " Size: " $1}' + } + + # DreamZero-DROID 14B warm-start checkpoint + download_model "GEAR-Dreams/DreamZero-DROID" "/fsx/models/DreamZero-DROID" + + # umt5-xxl text tokenizer / encoder + download_model "google/umt5-xxl" "/fsx/models/umt5-xxl" + + # LIBERO trajectories (LeRobot layout). + # MUST be physical-intelligence/libero (state/action schema matches + # the libero_sim preset), NOT lerobot/libero (different schema). + download_dataset "physical-intelligence/libero" "/fsx/datasets/libero" + + echo "" + echo "=== staging complete ===" + echo "Models:" + du -sh /fsx/models/DreamZero-DROID 2>/dev/null || echo " (not found)" + du -sh /fsx/models/umt5-xxl 2>/dev/null || echo " (not found)" + echo "Dataset:" + du -sh /fsx/datasets/libero 2>/dev/null || echo " (not found)" + ls -la /fsx/models/DreamZero-DROID /fsx/models/umt5-xxl /fsx/datasets/libero + resources: + requests: + cpu: "4" + memory: "8Gi" + limits: + cpu: "8" + memory: "16Gi" + volumeMounts: + - name: fsx + mountPath: /fsx + volumes: + - name: fsx + persistentVolumeClaim: + claimName: fsx-claim From 8868dbe3cef9dd33273c9a8466414ada09e97aa4 Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Wed, 17 Jun 2026 12:03:47 -0500 Subject: [PATCH 04/45] fix(dreamzero): update stale parent-repo path refs in script/config comments --- .../libero/scripts/libero_spatial_eval_dreamzero_14b.yaml | 4 ++-- .../kubernetes/libero/scripts/run_dreamzero_sft_eks.sh | 2 +- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/libero_spatial_eval_dreamzero_14b.yaml b/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/libero_spatial_eval_dreamzero_14b.yaml index 31c8942a6..ea54019a4 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/libero_spatial_eval_dreamzero_14b.yaml +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/libero_spatial_eval_dreamzero_14b.yaml @@ -31,12 +31,12 @@ # so the `model/dreamzero_14b` default must be 14B-correct on its own. # # HOW THIS FILE IS DELIVERED TO THE CONTAINER: -# This file lives in OUR repo (examples/dreamzero/configs/), NOT upstream. At +# This file lives in OUR repo (kubernetes/libero/scripts/), NOT upstream. At # runtime it is mounted into the eval pod via the `dreamzero-eval-config` # ConfigMap at /opt/eval-config/, and run_dreamzero_eval_eks.sh copies it into # /workspace/RLinf/examples/embodiment/config/ before invoking the eval (so # `--config-path config` finds it WITHOUT hiding the upstream env/ and model/ -# config groups it depends on). See examples/dreamzero/manifests/dreamzero-eval.yaml. +# config groups it depends on). See kubernetes/libero/dreamzero-eval.yaml. # # DROID action space: 8D (7 joint + gripper); model outputs action chunks. # ============================================================================= diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/run_dreamzero_sft_eks.sh b/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/run_dreamzero_sft_eks.sh index c9d0254bd..1a466845a 100755 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/run_dreamzero_sft_eks.sh +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/run_dreamzero_sft_eks.sh @@ -103,7 +103,7 @@ HYDRA_ARGS="${HYDRA_ARGS} runner.logger.log_path=${LOG_DIR}" # NOTE on checkpoint format for eval: # RLinf's FSDP saver writes a sharded DCP checkpoint (.distcp + .metadata) under # {ckpt}/actor/dcp_checkpoint/. The LIBERO eval needs a single .pt; convert the -# DCP to .pt offline (CPU) with examples/dreamzero/manifests/convert-checkpoint.yaml. +# DCP to .pt offline (CPU) with kubernetes/libero/convert-checkpoint.yaml. # We do NOT use +actor.fsdp_config.save_full_model_weights=true: on 2x p5en the # rank-0 full-state-dict gather for the 16B model is pathologically slow / stalls. # DCP-only save + offline convert is faster and more reliable. From 61c9c1532595b939cc1b85094dccdda72b334fca Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Wed, 17 Jun 2026 12:05:15 -0500 Subject: [PATCH 05/45] feat(dreamzero): add optional FSx storage, HF secret, and env_vars templates --- .../kubernetes/libero/env_vars.example | 18 +++++++++ .../kubernetes/libero/secret.example.yaml | 15 +++++++ .../libero/storage/pv-fsx-lustre-static.yaml | 39 +++++++++++++++++++ .../storage/pvc-fsx-lustre-dynamic.yaml | 33 ++++++++++++++++ 4 files changed, 105 insertions(+) create mode 100644 3.test_cases/pytorch/dreamzero/kubernetes/libero/env_vars.example create mode 100644 3.test_cases/pytorch/dreamzero/kubernetes/libero/secret.example.yaml create mode 100644 3.test_cases/pytorch/dreamzero/kubernetes/libero/storage/pv-fsx-lustre-static.yaml create mode 100644 3.test_cases/pytorch/dreamzero/kubernetes/libero/storage/pvc-fsx-lustre-dynamic.yaml diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/env_vars.example b/3.test_cases/pytorch/dreamzero/kubernetes/libero/env_vars.example new file mode 100644 index 000000000..6d6bebc7b --- /dev/null +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/env_vars.example @@ -0,0 +1,18 @@ +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 +# +# Copy to env_vars and `source env_vars` before running the walkthrough. +# env_vars is gitignored; env_vars.example is tracked. + +# Your ECR repository URI for the DreamZero training image (no tag). +export ECR_URI=.dkr.ecr..amazonaws.com/dreamzero + +# Kubernetes namespace to deploy into. +export NAMESPACE=dreamzero + +# AWS region of your cluster/ECR. +export AWS_REGION=us-east-1 + +# Pinned source refs (baked into the image build). +export UPSTREAM_REF=b3bbabb1f461 +export DREAMZERO_REF=ab790c198fbc diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/secret.example.yaml b/3.test_cases/pytorch/dreamzero/kubernetes/libero/secret.example.yaml new file mode 100644 index 000000000..9dd8767b0 --- /dev/null +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/secret.example.yaml @@ -0,0 +1,15 @@ +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 +# +# OPTIONAL: only needed if you must authenticate to gated HuggingFace repos. +# The default DreamZero/umt5/libero repos are public and download anonymously. +# Create with: kubectl -n ${NAMESPACE} create secret generic hf-token \ +# --from-literal=HF_TOKEN=hf_xxx +apiVersion: v1 +kind: Secret +metadata: + name: hf-token + namespace: ${NAMESPACE} +type: Opaque +stringData: + HF_TOKEN: "REPLACE_WITH_YOUR_HF_TOKEN" diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/storage/pv-fsx-lustre-static.yaml b/3.test_cases/pytorch/dreamzero/kubernetes/libero/storage/pv-fsx-lustre-static.yaml new file mode 100644 index 000000000..ff53f6d74 --- /dev/null +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/storage/pv-fsx-lustre-static.yaml @@ -0,0 +1,39 @@ +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 +# +# Optional: bind to an EXISTING FSx for Lustre filesystem (static provisioning). +# Fill in fileSystemId / dnsName / mountName from your FSx filesystem. The PVC is +# named fsx-claim (mounted at /fsx by the workflow manifests). +apiVersion: v1 +kind: PersistentVolume +metadata: + name: fsx-pv-dreamzero +spec: + capacity: + storage: 1200Gi + volumeMode: Filesystem + accessModes: + - ReadWriteMany + mountOptions: + - flock + persistentVolumeReclaimPolicy: Retain + csi: + driver: fsx.csi.aws.com + volumeHandle: ${FSX_FILESYSTEM_ID} + volumeAttributes: + dnsname: ${FSX_DNS_NAME} + mountname: ${FSX_MOUNT_NAME} +--- +apiVersion: v1 +kind: PersistentVolumeClaim +metadata: + name: fsx-claim + namespace: ${NAMESPACE} +spec: + accessModes: + - ReadWriteMany + storageClassName: "" + volumeName: fsx-pv-dreamzero + resources: + requests: + storage: 1200Gi diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/storage/pvc-fsx-lustre-dynamic.yaml b/3.test_cases/pytorch/dreamzero/kubernetes/libero/storage/pvc-fsx-lustre-dynamic.yaml new file mode 100644 index 000000000..e839df599 --- /dev/null +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/storage/pvc-fsx-lustre-dynamic.yaml @@ -0,0 +1,33 @@ +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 +# +# Optional: dynamically provision an FSx for Lustre filesystem + PVC named +# fsx-claim (mounted at /fsx by the workflow manifests). Requires the FSx CSI +# driver installed on the cluster. Size for >=250GB free (14B DCP checkpoint is +# ~140-206GB). Adjust subnetId/securityGroupIds to your cluster's VPC. +apiVersion: storage.k8s.io/v1 +kind: StorageClass +metadata: + name: fsx-sc +provisioner: fsx.csi.aws.com +parameters: + subnetId: ${FSX_SUBNET_ID} + securityGroupIds: ${FSX_SECURITY_GROUP_IDS} + deploymentType: SCRATCH_2 +mountOptions: + - flock +reclaimPolicy: Delete +volumeBindingMode: Immediate +--- +apiVersion: v1 +kind: PersistentVolumeClaim +metadata: + name: fsx-claim + namespace: ${NAMESPACE} +spec: + accessModes: + - ReadWriteMany + storageClassName: fsx-sc + resources: + requests: + storage: 1200Gi From 0de6ea74df576967757c32f5319e356f50e833bd Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Wed, 17 Jun 2026 12:07:00 -0500 Subject: [PATCH 06/45] feat(dreamzero): add docker buildx build-push helper --- 3.test_cases/pytorch/dreamzero/.gitignore | 3 ++ .../kubernetes/libero/setup/build-push.sh | 49 +++++++++++++++++++ 2 files changed, 52 insertions(+) create mode 100644 3.test_cases/pytorch/dreamzero/.gitignore create mode 100755 3.test_cases/pytorch/dreamzero/kubernetes/libero/setup/build-push.sh diff --git a/3.test_cases/pytorch/dreamzero/.gitignore b/3.test_cases/pytorch/dreamzero/.gitignore new file mode 100644 index 000000000..31866ae0c --- /dev/null +++ b/3.test_cases/pytorch/dreamzero/.gitignore @@ -0,0 +1,3 @@ +# Transient build clones created by kubernetes/libero/setup/build-push.sh +/RLinf/ +/DreamZero/ diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/setup/build-push.sh b/3.test_cases/pytorch/dreamzero/kubernetes/libero/setup/build-push.sh new file mode 100755 index 000000000..0a1b6130a --- /dev/null +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/setup/build-push.sh @@ -0,0 +1,49 @@ +#!/usr/bin/env bash +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 +# +# Build the DreamZero training image (two-stage: upstream RLinf embodied-libero + +# EFA overlay) and push to your ECR. Requires docker buildx + AWS CLI logged in. +# +# Usage: +# source ../env_vars # sets ECR_URI, AWS_REGION, UPSTREAM_REF, DREAMZERO_REF +# ./build-push.sh +set -euo pipefail + +: "${ECR_URI:?set ECR_URI (e.g. .dkr.ecr..amazonaws.com/dreamzero)}" +: "${AWS_REGION:?set AWS_REGION}" +UPSTREAM_REPO="${UPSTREAM_REPO:-https://github.com/RLinf/RLinf.git}" +UPSTREAM_REF="${UPSTREAM_REF:-b3bbabb1f461}" +DREAMZERO_REPO="${DREAMZERO_REPO:-https://github.com/RLinf/dreamzero.git}" +DREAMZERO_REF="${DREAMZERO_REF:-ab790c198fbc}" +BUILD_TARGET="${BUILD_TARGET:-embodied-libero}" +TAG="${TAG:-latest}" + +# Test-case root (this script is at kubernetes/libero/setup/build-push.sh). +ROOT="$(cd "$(dirname "$0")/../../.." && pwd)" +cd "$ROOT" + +echo "== ECR login ==" +aws ecr get-login-password --region "$AWS_REGION" | \ + docker login --username AWS --password-stdin "${ECR_URI%%/*}" + +echo "== Clone pinned sources into the build context ==" +rm -rf RLinf DreamZero +git clone "$UPSTREAM_REPO" RLinf && git -C RLinf checkout "$UPSTREAM_REF" +git clone "$DREAMZERO_REPO" DreamZero && git -C DreamZero checkout "$DREAMZERO_REF" + +echo "== Stage 1: upstream embodied-libero ==" +docker buildx build --platform linux/amd64 --load \ + --build-arg BUILD_TARGET="$BUILD_TARGET" --build-arg NO_MIRROR=1 \ + -t "rlinf-upstream-${BUILD_TARGET}" \ + -f RLinf/docker/Dockerfile RLinf + +echo "== Stage 2: EFA overlay (push) ==" +docker buildx build --platform linux/amd64 --push \ + --build-arg BUILD_TARGET="$BUILD_TARGET" \ + -t "${ECR_URI}:${TAG}" \ + -f Dockerfile . + +echo "== Done: ${ECR_URI}:${TAG} ==" +echo "== Cleaning transient clones ==" +rm -rf RLinf DreamZero From 2a138411d48515a68030adf3b14e89541d0f8cbf Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Wed, 17 Jun 2026 12:13:15 -0500 Subject: [PATCH 07/45] feat(dreamzero): add kaniko in-cluster build alternative (pending live validation) Parameterize the Dockerfile stage-2 base (RLINF_UPSTREAM_IMAGE) so the kaniko two-stage flow can FROM the ECR-pushed stage-1 tag. The docker buildx path (build-push.sh) remains the validated primary. The kaniko path is structurally complete and renders valid, but the in-cluster build has NOT yet been run end-to-end -- README marks it pending live validation. --- 3.test_cases/pytorch/dreamzero/Dockerfile | 6 +- .../kubernetes/libero/setup/kaniko-build.yaml | 133 ++++++++++++++++++ 2 files changed, 138 insertions(+), 1 deletion(-) create mode 100644 3.test_cases/pytorch/dreamzero/kubernetes/libero/setup/kaniko-build.yaml diff --git a/3.test_cases/pytorch/dreamzero/Dockerfile b/3.test_cases/pytorch/dreamzero/Dockerfile index 18b502a29..4134d86f2 100644 --- a/3.test_cases/pytorch/dreamzero/Dockerfile +++ b/3.test_cases/pytorch/dreamzero/Dockerfile @@ -44,7 +44,11 @@ FROM rlinf-upstream-${BUILD_TARGET} AS upstream # Layer EFA, GDRCopy, NCCL, and OpenMPI onto the upstream RLinf image. # Recipe proven in nccl-tests/Dockerfile on nvidia/cuda base images. -FROM rlinf-upstream-${BUILD_TARGET} +# RLINF_UPSTREAM_IMAGE lets the kaniko build path (setup/kaniko-build.yaml) point +# stage 2 at the stage-1 image it pushed to ECR. The docker buildx path +# (setup/build-push.sh) uses the default local tag built immediately before. +ARG RLINF_UPSTREAM_IMAGE=rlinf-upstream-${BUILD_TARGET} +FROM ${RLINF_UPSTREAM_IMAGE} ARG GDRCOPY_VERSION=v2.5.1 ARG EFA_INSTALLER_VERSION=1.47.0 diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/setup/kaniko-build.yaml b/3.test_cases/pytorch/dreamzero/kubernetes/libero/setup/kaniko-build.yaml new file mode 100644 index 000000000..4674e0862 --- /dev/null +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/setup/kaniko-build.yaml @@ -0,0 +1,133 @@ +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 +# +# In-cluster image build via the Chainguard kaniko fork (no local Docker daemon +# or disk needed). Alternative to setup/build-push.sh for disk-constrained users. +# +# Two stages, run as two Jobs (apply stage 1, wait for completion, then stage 2): +# stage 1: build the upstream RLinf embodied-libero image, push as +# ${ECR_URI}:upstream-embodied-libero +# stage 2: FROM that tag, layer the EFA overlay (this test case's Dockerfile), +# apply the DCP-save patch, push ${ECR_URI}:latest +# +# A shared FSx scratch dir (PVC fsx-claim) holds the cloned sources + build cache. +# The test-case Dockerfile + docker/ build helpers are delivered via a ConfigMap +# (created from the test-case dir; see kubernetes/libero/README.md). +# +# Apply (restricted envsubst): +# kubectl -n ${NAMESPACE} create configmap dreamzero-build-context \ +# --from-file=Dockerfile=Dockerfile \ +# --from-file=install_extras.sh=docker/scripts/install_extras.sh \ +# --from-file=dcp-save-finalize-besteffort.patch=docker/scripts/patches/dcp-save-finalize-besteffort.patch \ +# --dry-run=client -o yaml | kubectl apply -f - +# envsubst '${ECR_URI} ${NAMESPACE}' < kubernetes/libero/setup/kaniko-build.yaml | kubectl apply -f - +--- +apiVersion: batch/v1 +kind: Job +metadata: + name: dreamzero-kaniko-stage1 + namespace: ${NAMESPACE} + labels: + app: dreamzero-kaniko + stage: "1" +spec: + backoffLimit: 0 + template: + metadata: + labels: + app: dreamzero-kaniko + stage: "1" + spec: + restartPolicy: Never + serviceAccountName: training-sa + initContainers: + - name: clone-rlinf + image: alpine/git:2.45.2 + command: ["sh", "-c"] + args: + - | + set -eux + rm -rf /ctx/RLinf + git clone https://github.com/RLinf/RLinf.git /ctx/RLinf + git -C /ctx/RLinf checkout b3bbabb1f461 + volumeMounts: + - { name: ctx, mountPath: /ctx } + containers: + - name: kaniko + image: ghcr.io/chainguard-forks/kaniko/executor:latest + args: + - --dockerfile=/ctx/RLinf/docker/Dockerfile + - --context=dir:///ctx/RLinf + - --build-arg=BUILD_TARGET=embodied-libero + - --build-arg=NO_MIRROR=1 + - --destination=${ECR_URI}:upstream-embodied-libero + - --cache=true + - --cache-repo=${ECR_URI}/cache + volumeMounts: + - { name: ctx, mountPath: /ctx } + volumes: + - name: ctx + persistentVolumeClaim: + claimName: fsx-claim +--- +apiVersion: batch/v1 +kind: Job +metadata: + name: dreamzero-kaniko-stage2 + namespace: ${NAMESPACE} + labels: + app: dreamzero-kaniko + stage: "2" +spec: + backoffLimit: 0 + template: + metadata: + labels: + app: dreamzero-kaniko + stage: "2" + spec: + restartPolicy: Never + serviceAccountName: training-sa + initContainers: + - name: assemble-context + image: alpine/git:2.45.2 + command: ["sh", "-c"] + args: + - | + set -eux + # Stage-2 build context = /ctx/overlay containing the test-case + # Dockerfile + docker/scripts/ (helpers + DCP patch) + the DreamZero + # groot clone. RLinf source is re-cloned for the COPY RLinf/ layer. + rm -rf /ctx/overlay + mkdir -p /ctx/overlay/docker/scripts/patches + cp /cm/Dockerfile /ctx/overlay/Dockerfile + cp /cm/install_extras.sh /ctx/overlay/docker/scripts/install_extras.sh + cp /cm/dcp-save-finalize-besteffort.patch /ctx/overlay/docker/scripts/patches/dcp-save-finalize-besteffort.patch + git clone https://github.com/RLinf/RLinf.git /ctx/overlay/RLinf + git -C /ctx/overlay/RLinf checkout b3bbabb1f461 + git clone https://github.com/RLinf/dreamzero.git /ctx/overlay/DreamZero + git -C /ctx/overlay/DreamZero checkout ab790c198fbc + volumeMounts: + - { name: ctx, mountPath: /ctx } + - { name: cm, mountPath: /cm } + containers: + - name: kaniko + image: ghcr.io/chainguard-forks/kaniko/executor:latest + args: + - --dockerfile=/ctx/overlay/Dockerfile + - --context=dir:///ctx/overlay + - --build-arg=BUILD_TARGET=embodied-libero + - --build-arg=RLINF_UPSTREAM_IMAGE=${ECR_URI}:upstream-embodied-libero + - --destination=${ECR_URI}:latest + - --cache=true + - --cache-repo=${ECR_URI}/cache + volumeMounts: + - { name: ctx, mountPath: /ctx } + volumes: + - name: ctx + persistentVolumeClaim: + claimName: fsx-claim + - name: cm + configMap: + name: dreamzero-build-context + optional: false From 4c5772edb925df9f53b83d9fe56beba55dfc376f Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Wed, 17 Jun 2026 12:14:49 -0500 Subject: [PATCH 08/45] feat(dreamzero): add diagrams (incl. rendered infra SVG) + assets via Git LFS WAM training/inference + infra-topology draw.io sources and rendered SVGs; rollout mp4 + loss-curve png. Binaries tracked via Git LFS (.gitattributes). Rendered the previously-missing infra-dreamzero-sft SVG with the draw.io CLI. --- 3.test_cases/pytorch/dreamzero/.gitattributes | 3 + .../dreamzero/assets/libero-rollout-seed0.mp4 | 3 + .../pytorch/dreamzero/assets/loss-curve.png | 3 + .../diagrams/dreamzero-wam-inference.drawio | 144 +++++++++++++ .../dreamzero-wam-inference.drawio.svg | 3 + .../dreamzero/diagrams/dreamzero-wam.drawio | 193 ++++++++++++++++++ .../diagrams/dreamzero-wam.drawio.svg | 3 + .../diagrams/infra-dreamzero-sft.drawio | 86 ++++++++ .../diagrams/infra-dreamzero-sft.drawio.svg | 3 + 9 files changed, 441 insertions(+) create mode 100644 3.test_cases/pytorch/dreamzero/.gitattributes create mode 100644 3.test_cases/pytorch/dreamzero/assets/libero-rollout-seed0.mp4 create mode 100644 3.test_cases/pytorch/dreamzero/assets/loss-curve.png create mode 100644 3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam-inference.drawio create mode 100644 3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam-inference.drawio.svg create mode 100644 3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam.drawio create mode 100644 3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam.drawio.svg create mode 100644 3.test_cases/pytorch/dreamzero/diagrams/infra-dreamzero-sft.drawio create mode 100644 3.test_cases/pytorch/dreamzero/diagrams/infra-dreamzero-sft.drawio.svg diff --git a/3.test_cases/pytorch/dreamzero/.gitattributes b/3.test_cases/pytorch/dreamzero/.gitattributes new file mode 100644 index 000000000..f416907f4 --- /dev/null +++ b/3.test_cases/pytorch/dreamzero/.gitattributes @@ -0,0 +1,3 @@ +assets/*.mp4 filter=lfs diff=lfs merge=lfs -text +assets/*.png filter=lfs diff=lfs merge=lfs -text +diagrams/*.svg filter=lfs diff=lfs merge=lfs -text diff --git a/3.test_cases/pytorch/dreamzero/assets/libero-rollout-seed0.mp4 b/3.test_cases/pytorch/dreamzero/assets/libero-rollout-seed0.mp4 new file mode 100644 index 000000000..81dcfa537 --- /dev/null +++ b/3.test_cases/pytorch/dreamzero/assets/libero-rollout-seed0.mp4 @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b8a735fe4ac0185deb56ed54054e414a165376d042b662fe3fd283cb058630cc +size 61217 diff --git a/3.test_cases/pytorch/dreamzero/assets/loss-curve.png b/3.test_cases/pytorch/dreamzero/assets/loss-curve.png new file mode 100644 index 000000000..8e46c78df --- /dev/null +++ b/3.test_cases/pytorch/dreamzero/assets/loss-curve.png @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6c99d97ce8e928af261fe2cd1ae9dba45c3e9d978062fbc57d76b21effe039a2 +size 147648 diff --git a/3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam-inference.drawio b/3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam-inference.drawio new file mode 100644 index 000000000..9dd13aaa7 --- /dev/null +++ b/3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam-inference.drawio @@ -0,0 +1,144 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam-inference.drawio.svg b/3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam-inference.drawio.svg new file mode 100644 index 000000000..71c127d81 --- /dev/null +++ b/3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam-inference.drawio.svg @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:12cab47be2b07cfd3e13db1f54293b5e68ec5e4ed1fd250dfc4cca74f28d2f31 +size 732106 diff --git a/3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam.drawio b/3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam.drawio new file mode 100644 index 000000000..3b4402961 --- /dev/null +++ b/3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam.drawio @@ -0,0 +1,193 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam.drawio.svg b/3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam.drawio.svg new file mode 100644 index 000000000..f944e7101 --- /dev/null +++ b/3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam.drawio.svg @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2a2190a15bbe23ea91b993b73f5d0b2a8a434b42846ba323dafb60fbabc5643d +size 716235 diff --git a/3.test_cases/pytorch/dreamzero/diagrams/infra-dreamzero-sft.drawio b/3.test_cases/pytorch/dreamzero/diagrams/infra-dreamzero-sft.drawio new file mode 100644 index 000000000..10cb08540 --- /dev/null +++ b/3.test_cases/pytorch/dreamzero/diagrams/infra-dreamzero-sft.drawio @@ -0,0 +1,86 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/3.test_cases/pytorch/dreamzero/diagrams/infra-dreamzero-sft.drawio.svg b/3.test_cases/pytorch/dreamzero/diagrams/infra-dreamzero-sft.drawio.svg new file mode 100644 index 000000000..a1686b168 --- /dev/null +++ b/3.test_cases/pytorch/dreamzero/diagrams/infra-dreamzero-sft.drawio.svg @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b17ad4438637ffb367e61c975739b313ecd03f47950ed9f97c443e821a8a4c3b +size 761669 From 813952810570011eb5d13f9fd3944c1e6d9fbbeb Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Wed, 17 Jun 2026 12:21:08 -0500 Subject: [PATCH 09/45] docs(dreamzero): add kubernetes/libero walkthrough README --- .../dreamzero/kubernetes/libero/README.md | 542 ++++++++++++++++++ 1 file changed, 542 insertions(+) create mode 100644 3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md new file mode 100644 index 000000000..65fdd90b5 --- /dev/null +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md @@ -0,0 +1,542 @@ + + + +# Continue SFT of DreamZero (14B World-Action Model) on LIBERO on Amazon EKS + +DreamZero is a **16.48B-parameter World-Action Model (WAM)** — a Wan-based +video-diffusion Diffusion Transformer (DiT) that *jointly* denoises future video +frames and future robot actions in a shared causal self-attention space. The +model predicts both what will happen (video) and what to do (actions); the video +prediction acts as a computational scaffold for action reasoning. + +This walkthrough packages the canonical customer workflow as a set of reusable +Kubernetes manifests that run on Amazon EKS: take the released +[`GEAR-Dreams/DreamZero-DROID`](https://huggingface.co/GEAR-Dreams/DreamZero-DROID) +14B checkpoint (pretrained on DROID, a Franka arm), continue **supervised +fine-tuning (SFT)** on a *new* embodiment's data (LIBERO, `libero_sim`), then +**evaluate the result in the LIBERO simulator** and render in-sim rollout videos. +There is no native LIBERO 14B checkpoint upstream — warm-starting from DROID is +the point. + +Upstream projects: +[github.com/RLinf/RLinf](https://github.com/RLinf/RLinf) (training framework) and +[github.com/RLinf/dreamzero](https://github.com/RLinf/dreamzero) (the `groot` +package that provides the WAM model code). + +> **Validation scope (read this first).** The pipeline below was validated +> end-to-end on EKS with a **1-step** SFT run. That proves the *infrastructure +> and the pipeline* — image build, multi-node EFA/NCCL, FSDP2 sharded +> checkpointing, DCP→`.pt` conversion, and LIBERO simulator eval — **not** task +> accuracy. A 1-step checkpoint yields `eval/success_once = 0.0`, which is +> expected. Real accuracy requires a multi-step training run (the pipeline +> supports it — raise `runner.max_steps`; see step 4). No success numbers or loss +> curves are fabricated here. + +The recipe has six moving parts, each a self-contained manifest in this +directory: + +1. A one-shot **model/dataset download Job** (`model-download.yaml`) that stages + the DreamZero-DROID checkpoint, the `umt5-xxl` tokenizer, and the + `physical-intelligence/libero` dataset onto a shared FSx for Lustre volume. +2. A **metadata-generation Job** (`generate-metadata.yaml`) that produces the + `libero_sim` normalization statistics the DROID checkpoint does not ship. +3. A **multi-node SFT RayJob** (`dreamzero-sft.yaml`) that fans FSDP2 training + across 2× `p5en.48xlarge` (16× H200) via KubeRay. +4. A **checkpoint-conversion Job** (`convert-checkpoint.yaml`) that consolidates + the sharded FSDP DCP checkpoint into a single `.pt` on CPU. +5. A **LIBERO simulator eval Job** (`dreamzero-eval.yaml`) that drives the sim, + reports `eval/success_once`, and writes in-sim rollout videos. + +All steps share a single FSx for Lustre `PersistentVolumeClaim` named +`fsx-claim`, mounted at `/fsx`, that holds the models, dataset, and checkpoints. + +## Architecture + +### Infrastructure / training topology + +![DreamZero SFT infrastructure](../../diagrams/infra-dreamzero-sft.drawio.svg) + +Multi-node SFT runs as a **KubeRay `RayJob`** (run-to-completion) with an embedded +`RayCluster`: one Ray **head** pod and one Ray **worker** pod, each landing on its +own `p5en.48xlarge` node (8× H200 141 GB) → **16 GPUs total**. The KubeRay +operator brings the Ray cluster up, then runs the Ray-agnostic launcher +(`run_dreamzero_sft_eks.sh`) as the head `entrypoint`; RLinf's `Cluster` (Ray) +scheduler fans **FSDP2 `full_shard`** training across all 16 GPUs (the 16.48B +model is *sharded*, not replicated). There is **no** torchrun, DeepSpeed, or +manual head election. Gradient sync flows over **NCCL on EFA RDMA** (libfabric +2.4 / aws-ofi-nccl 1.18, GPUDirect RDMA). Pod anti-affinity guarantees one pod +per physical node; a topology-spread constraint prefers co-location under the +same network layer for lowest NCCL latency. `shutdownAfterJobFinishes: true` +tears the RayCluster down when training ends. + +### The World-Action Model + +![DreamZero WAM](../../diagrams/dreamzero-wam.drawio.svg) + +During training, video frames and actions are each encoded and corrupted with +noise via flow-matching interpolation. The noisy video latent tokens and an +**Action Register** (noisy action tokens + state-encoder output) participate +together in blockwise causal self-attention — this is the "joint" in Joint +Video-Action. UMT5-XXL text embeddings and CLIP image embeddings condition the +DiT via cross-attention. Two output heads predict the velocity field: one for +video (dynamics loss) and one for actions (action loss), weighted equally. The +video loss is not auxiliary — it is the mechanism by which the model learns +physics (gravity, contact, object permanence). At deployment the robot consumes +only the action channel. + +| Component | Role | +|-----------|------| +| DiT backbone (14B-class WAM) | Shared denoising over video + action tokens (dim 5120, 40 layers) | +| Action Register | Noisy action tokens + state-encoder output; participates in joint attention | +| UMT5-XXL | Encodes task-instruction text (`google/umt5-xxl` tokenizer) | +| CLIP / image conditioning | Encodes the observation frame for cross-attention conditioning | +| Wan VAE | Compresses/decodes video latents | +| Action Decoder | Projects denoised action tokens to per-embodiment joint positions | + +## Hardware requirements + +| Resource | Requirement | Notes | +|----------|-------------|-------| +| GPU nodes | **2× `p5en.48xlarge`** | 8× NVIDIA H200 (141 GB) each = **16 GPUs**. The 16.48B model is FSDP2-sharded across all 16. | +| EFA | **16 EFA NICs per node** | High-bandwidth RDMA for NCCL allreduce. Single-node eval also requests 16. | +| Shared storage | **FSx for Lustre, ≥250 GB free** | A 14B FSDP **DCP checkpoint is ~140–206 GB** (16 shards incl. full optimizer state). A full filesystem truncates `torch.save` mid-write → corrupt, unreadable shards. | +| Eval | 1× `p5en.48xlarge` (8× H200) | Eval pins `cluster.num_nodes=1`; the 14B model is sharded across one node's 8 GPUs with single-node FSDP. | + +## Prerequisites + +### 1. An EKS cluster that can provision 2× `p5en.48xlarge` with EFA + +You need an Amazon EKS cluster with GPU autoscaling (e.g. Karpenter) able to +launch **2× `p5en.48xlarge`** nodes, each with **8× H200** GPUs and **16 EFA +NICs**, plus the NVIDIA GPU Operator (or device plugin) and EFA device plugin so +pods can request `nvidia.com/gpu` and `vpc.amazonaws.com/efa`. Cluster-creation +references live in +[`1.architectures/4.amazon-eks`](../../../../../1.architectures/4.amazon-eks). + +Point your local kubeconfig at the cluster and confirm it is reachable: + +```bash +aws eks update-kubeconfig --name --region +kubectl config current-context +``` + +### 2. KubeRay operator + +Multi-node SFT is a `ray.io/v1` `RayJob`, so the KubeRay operator must be +installed: + +```bash +helm repo add kuberay https://ray-project.github.io/kuberay-helm/ +helm install kuberay-operator kuberay/kuberay-operator \ + --version 1.6.0 \ + -n kuberay-system --create-namespace +``` + +Verify the operator is running: + +```bash +kubectl get pods -n kuberay-system +``` + +### 3. FSx for Lustre shared storage (`fsx-claim` at `/fsx`) + +Every step mounts a `PersistentVolumeClaim` named **`fsx-claim`** at `/fsx`. If +your cluster already exposes such a PVC (many EKS GPU cluster templates ship one), +confirm it is `Bound` with `ReadWriteMany` access and has ≥250 GB free: + +```bash +kubectl get pvc fsx-claim -n "$NAMESPACE" +# NAME STATUS VOLUME CAPACITY ACCESS MODES STORAGECLASS AGE +# fsx-claim Bound fsx-pv... 1.2Ti RWX fsx-sc 3d +``` + +If you do **not** already have one, this directory ships two optional manifests +(both require the `fsx.csi.aws.com` CSI driver): + +- **Dynamic provisioning** — `storage/pvc-fsx-lustre-dynamic.yaml` creates a + `StorageClass` + `fsx-claim` PVC and provisions a fresh filesystem. Fill in + `FSX_SUBNET_ID` and `FSX_SECURITY_GROUP_IDS` for your cluster's VPC: + + ```bash + export FSX_SUBNET_ID=subnet-0abc... FSX_SECURITY_GROUP_IDS=sg-0def... + envsubst < storage/pvc-fsx-lustre-dynamic.yaml | kubectl apply -f - + ``` + +- **Static binding** — `storage/pv-fsx-lustre-static.yaml` binds `fsx-claim` to + an existing FSx for Lustre filesystem. Fill in `FSX_FILESYSTEM_ID`, + `FSX_DNS_NAME`, and `FSX_MOUNT_NAME`: + + ```bash + export FSX_FILESYSTEM_ID=fs-0... FSX_DNS_NAME=fs-0....fsx..amazonaws.com FSX_MOUNT_NAME=abcd1234 + envsubst < storage/pv-fsx-lustre-static.yaml | kubectl apply -f - + ``` + +### 4. A `training-sa` ServiceAccount with credentials for HF + S3 + +The SFT, eval, convert, and metadata pods run as the **`training-sa`** +ServiceAccount. Give it IRSA or EKS Pod Identity so pods can reach Hugging Face +and (if you stage to/from S3) Amazon S3. For example, with IRSA via `eksctl`: + +```bash +eksctl create iamserviceaccount \ + --name training-sa \ + --namespace "$NAMESPACE" \ + --cluster \ + --attach-policy-arn arn:aws:iam::aws:policy/AmazonS3ReadOnlyAccess \ + --approve +``` + +Or attach a Pod Identity association to an existing ServiceAccount: + +```bash +kubectl create serviceaccount training-sa -n "$NAMESPACE" +aws eks create-pod-identity-association \ + --cluster-name \ + --namespace "$NAMESPACE" \ + --service-account training-sa \ + --role-arn arn:aws:iam:::role/ +``` + +> **Hugging Face auth is optional.** The default repos +> (`GEAR-Dreams/DreamZero-DROID`, `google/umt5-xxl`, +> `physical-intelligence/libero`) are **public** and download anonymously. Only +> if you must authenticate to a gated repo, create the `hf-token` Secret from +> `secret.example.yaml`: +> +> ```bash +> kubectl -n "$NAMESPACE" create secret generic hf-token --from-literal=HF_TOKEN=hf_xxx +> ``` + +### 5. Local tooling: `kubectl`, `helm`, and the `a8m/envsubst` variant + +Several manifests in this directory embed inline shell scripts inside the YAML +(the RayJob `entrypoint`, the metadata bootstrap, the convert and eval launchers) +and are rendered with a **restricted** `envsubst` allow-list so only +`${ECR_URI}` and `${NAMESPACE}` are substituted while inline shell variables +(`${PYTHONPATH:-}`, `${DREAMZERO_PATH}`, …) are left literal. + +Install [a8m/envsubst](https://github.com/a8m/envsubst) — **not** GNU gettext's +`envsubst`, which does not support the same allow-list/escape semantics: + +```bash +# macOS / Linux prebuilt binary (uname picks the right asset): +curl -L "https://github.com/a8m/envsubst/releases/download/v1.4.3/envsubst-$(uname -s)-$(uname -m)" \ + -o /usr/local/bin/envsubst +chmod +x /usr/local/bin/envsubst + +# Or, with Go installed: +go install github.com/a8m/envsubst/cmd/envsubst@latest +``` + +Verify the binary on your PATH is the a8m one: + +```bash +envsubst --version +# expect: "envsubst version: vX.Y.Z" (a8m/envsubst) +# if you see "envsubst (GNU gettext-runtime)", the GNU binary is still on PATH. +``` + +## Configure + +The committed template is `env_vars.example`. Copy it to `env_vars` (gitignored), +edit your values, then `source` it. Every step below assumes you have sourced it: + +```bash +cp env_vars.example env_vars +"${EDITOR:-vi}" env_vars +source env_vars +``` + +The variables: + +| Variable | Example | Purpose | +|----------|---------|---------| +| `ECR_URI` | `.dkr.ecr..amazonaws.com/dreamzero` | Your ECR repository for the DreamZero training image (no tag). | +| `NAMESPACE` | `dreamzero` | Kubernetes namespace to deploy into. | +| `AWS_REGION` | `us-east-1` | Region of your cluster / ECR. | +| `UPSTREAM_REF` | `b3bbabb1f461` | Pinned RLinf commit baked into the image. | +| `DREAMZERO_REF` | `ab790c198fbc` | Pinned `dreamzero` (`groot`) commit baked into the image. | + +## Step-by-step + +Throughout, `$NAMESPACE` and `$ECR_URI` come from `env_vars`. Wherever a manifest +embeds inline shell, render it with the **restricted** allow-list +`envsubst '${ECR_URI} ${NAMESPACE}'` so the inline `${...}` shell variables are +not clobbered to empty strings. + +### 1. Build and push the training image + +The image is built in two stages: the upstream RLinf `embodied-libero` target +(which natively builds the dedicated **`dreamzero` venv**, RLinf PR #1272), then +an EFA overlay (`Dockerfile` at the test-case root) that layers EFA 1.47.0 / +libfabric 2.4.0 / aws-ofi-nccl 1.18.0 and a best-effort DCP-save patch. The +external `groot` package is cloned from `github.com/RLinf/dreamzero.git` and +placed on `PYTHONPATH` via `DREAMZERO_PATH=/workspace/DreamZero`. + +**Primary path (local Docker + buildx).** Run from the `setup/` directory; it +reads `ECR_URI`, `AWS_REGION`, `UPSTREAM_REF`, and `DREAMZERO_REF` from +`env_vars`: + +```bash +cd setup +source ../env_vars +./build-push.sh +cd .. +``` + +This clones the pinned `RLinf` and `dreamzero` sources, builds stage 1 +(`rlinf-upstream-embodied-libero`), then builds and pushes stage 2 to +`${ECR_URI}:latest`. + +**Alternative path (in-cluster kaniko, disk-constrained hosts).** + +> ⚠️ **EXPERIMENTAL — PENDING LIVE VALIDATION.** The kaniko in-cluster build path +> below is provided for hosts without a local Docker daemon or sufficient disk. +> It is **not-yet-validated** / **untested in-cluster as of this writing**. The +> primary `build-push.sh` path above is the validated one. Use kaniko at your own +> risk and expect to debug the two-stage build. + +The kaniko path (`setup/kaniko-build.yaml`) runs two Jobs against a shared FSx +scratch dir. First create the build-context ConfigMap from the test-case +`Dockerfile` and `docker/scripts/`, then apply the two stages in order (wait for +stage 1 to complete before stage 2): + +```bash +# From the test-case root (3.test_cases/pytorch/dreamzero), create the context CM: +kubectl -n "$NAMESPACE" create configmap dreamzero-build-context \ + --from-file=Dockerfile=Dockerfile \ + --from-file=install_extras.sh=docker/scripts/install_extras.sh \ + --from-file=dcp-save-finalize-besteffort.patch=docker/scripts/patches/dcp-save-finalize-besteffort.patch \ + --dry-run=client -o yaml | kubectl apply -f - + +# Apply both stages (restricted envsubst), then wait on stage 1 before stage 2 runs: +envsubst '${ECR_URI} ${NAMESPACE}' < kubernetes/libero/setup/kaniko-build.yaml | kubectl apply -f - +kubectl wait --for=condition=complete job/dreamzero-kaniko-stage1 -n "$NAMESPACE" --timeout=90m +kubectl logs -f job/dreamzero-kaniko-stage2 -n "$NAMESPACE" +``` + +### 2. Stage models + dataset to FSx + +Downloads the DreamZero-DROID 14B warm-start checkpoint +(`GEAR-Dreams/DreamZero-DROID`), the `google/umt5-xxl` tokenizer, and the +`physical-intelligence/libero` dataset (LeRobot layout) — all **anonymous**. This +Job runs in a lightweight `python:3.11-slim` staging image, so plain `envsubst` +is fine: + +```bash +envsubst < model-download.yaml | kubectl apply -f - +kubectl logs -f -n "$NAMESPACE" job/model-download-dreamzero +``` + +> **The dataset MUST be `physical-intelligence/libero`, NOT `lerobot/libero`.** +> The two repos share a name but have *different* `observation.state` / `action` +> column schemas. Only `physical-intelligence/libero` matches the `libero_sim` +> preset. Using `lerobot/libero` silently trains on the wrong column layout. + +Stages to `/fsx/models/DreamZero-DROID`, `/fsx/models/umt5-xxl`, and +`/fsx/datasets/libero` (~152 GB total). + +### 3. Generate the `libero_sim` normalization metadata + +The DreamZero-DROID checkpoint bundles `experiment_cfg/metadata.json` for +embodiment `oxe_droid` **only**, so LIBERO SFT (`embodiment_tag: libero_sim`) +would fail with `KeyError: embodiment_tag 'libero_sim' not found`. This Job runs +upstream's `toolkits/lerobot/generate_dreamzero_metadata.py` inside the **training +image** (it needs the RLinf toolkit + the `dreamzero` venv), so use **restricted** +`envsubst`: + +```bash +envsubst '${ECR_URI} ${NAMESPACE}' < generate-metadata.yaml | kubectl apply -f - +kubectl logs -f -n "$NAMESPACE" job/generate-metadata-dreamzero +# -> writes /fsx/models/metadata-libero.json (top-level key `libero_sim`) +``` + +The SFT and eval launchers default `METADATA_PATH` to this path, so downstream +steps pick it up automatically. + +### 4. Multi-node SFT (2× `p5en.48xlarge`) + +Create the launcher ConfigMap from `scripts/run_dreamzero_sft_eks.sh`, then apply +the RayJob with **restricted** `envsubst` (the RayJob `entrypoint` embeds an +inline `bash -c` block that unrestricted substitution would mangle): + +```bash +kubectl -n "$NAMESPACE" create configmap dreamzero-sft-launcher \ + --from-file=run_dreamzero_sft_eks.sh=scripts/run_dreamzero_sft_eks.sh \ + --dry-run=client -o yaml | kubectl apply -f - + +envsubst '${ECR_URI} ${NAMESPACE}' < dreamzero-sft.yaml | kubectl apply -f - + +kubectl logs -f -n "$NAMESPACE" job/dreamzero-sft +``` + +The launcher drives `examples/sft/train_vla_sft.py` with config +`libero_sft_dreamzero_14b`. SFT writes a **sharded FSDP DCP** checkpoint +(`.distcp` + `.metadata`) under +`.../global_step_/actor/dcp_checkpoint/`. + +> **Validated smoke run uses `runner.max_steps=1`.** For a real (multi-step) +> training run, raise `runner.max_steps` (and set a checkpoint `save_interval`) +> via the launcher's `HYDRA_OVERRIDES` env var on the RayJob head/worker +> containers, e.g. `HYDRA_OVERRIDES="runner.max_steps=2000 runner.save_interval=500"`. + +### 5. Convert the checkpoint (DCP shards → single `.pt`) + +Eval consumes a single consolidated `.pt`. Convert the sharded DCP offline on +**CPU** (do **not** use `save_full_model_weights` — the rank-0 full-state-dict +gather stalls on the 16B model). Create the launcher ConfigMap from +`scripts/convert_checkpoint.sh`, then apply with **restricted** `envsubst`: + +```bash +kubectl -n "$NAMESPACE" create configmap dreamzero-convert-launcher \ + --from-file=convert_checkpoint.sh=scripts/convert_checkpoint.sh \ + --dry-run=client -o yaml | kubectl apply -f - + +envsubst '${ECR_URI} ${NAMESPACE}' < convert-checkpoint.yaml | kubectl apply -f - +kubectl logs -f -n "$NAMESPACE" job/dreamzero-convert +# -> .../global_step_1/actor/model_state_dict/full_weights.pt +``` + +The default `STEP=global_step_1` matches the 1-step smoke run; override the +`STEP` env in the manifest for a later checkpoint. + +### 6. LIBERO simulator eval (single-node GPU) + +Single-pod GPU Job (`cluster.num_nodes=1`, single-node FSDP across 8 H200). Uses +the 14B eval config `scripts/libero_spatial_eval_dreamzero_14b.yaml` (upstream +ships only a 5B eval config). Create **both** ConfigMaps — the launcher *and* the +14B eval config — then apply with **restricted** `envsubst`: + +```bash +kubectl -n "$NAMESPACE" create configmap dreamzero-eval-launcher \ + --from-file=run_dreamzero_eval_eks.sh=scripts/run_dreamzero_eval_eks.sh \ + --dry-run=client -o yaml | kubectl apply -f - + +kubectl -n "$NAMESPACE" create configmap dreamzero-eval-config \ + --from-file=libero_spatial_eval_dreamzero_14b.yaml=scripts/libero_spatial_eval_dreamzero_14b.yaml \ + --dry-run=client -o yaml | kubectl apply -f - + +envsubst '${ECR_URI} ${NAMESPACE}' < dreamzero-eval.yaml | kubectl apply -f - +kubectl logs -f -n "$NAMESPACE" job/dreamzero-eval +``` + +The launcher copies the eval config into the embodiment config dir at runtime +(mounting a ConfigMap over the dir would hide upstream config groups). The eval +reports `eval/success_once` and (with `SAVE_VIDEO=True`, the default) writes +in-sim rollout videos to `{LOG_DIR}/video/eval/seed_*/0.mp4` +(`/fsx/checkpoints/dreamzero-libero-eval/video/eval/seed_*/0.mp4`). + +> The 14B config uses `total_num_envs=16`; the upstream 5B default of **128 OOMs** +> the 16.48B model co-located with the sim on 8× H200. +> +> **`eval/success_once = 0.0` is EXPECTED for a 1-step checkpoint** — the eval +> validates the machinery, not policy competence. + +## File structure + +``` +kubernetes/libero/ +├── README.md # This walkthrough +├── env_vars.example # Copy to env_vars and `source` it +├── secret.example.yaml # OPTIONAL hf-token Secret (gated repos only) +├── model-download.yaml # Job: stage DreamZero-DROID + umt5-xxl + libero dataset +├── generate-metadata.yaml # Job: libero_sim normalization metadata.json +├── dreamzero-sft.yaml # KubeRay RayJob: 2-node FSDP2 SFT +├── convert-checkpoint.yaml # Job (CPU): DCP shards -> full_weights.pt +├── dreamzero-eval.yaml # Job (GPU): LIBERO sim eval + in-sim video +├── scripts/ +│ ├── run_dreamzero_sft_eks.sh # Multi-node SFT launcher (Ray-agnostic, FSDP2) +│ ├── convert_checkpoint.sh # DCP -> .pt conversion launcher +│ ├── run_dreamzero_eval_eks.sh # LIBERO simulator eval launcher +│ └── libero_spatial_eval_dreamzero_14b.yaml # 14B eval config (upstream ships 5B only) +├── setup/ +│ ├── build-push.sh # Primary: local buildx two-stage build + push +│ └── kaniko-build.yaml # EXPERIMENTAL in-cluster kaniko build (pending validation) +└── storage/ + ├── pvc-fsx-lustre-dynamic.yaml # OPTIONAL: dynamically provision fsx-claim + └── pv-fsx-lustre-static.yaml # OPTIONAL: bind fsx-claim to an existing FSx +``` + +## Configuration deep-dive + +A few non-obvious config decisions are baked into the launchers and the 14B eval +config: + +- **`actor.model.num_action_per_block=16` (temporal alignment).** + `libero_sft_dreamzero_14b.yaml` sets `action_horizon=16` but inherits + `num_action_per_block=24` from the DROID model default. The mismatch trips the + forward-pass assertion + (`actions.shape[1] / (noise.shape[1]-1) == num_action_per_block // num_frame_per_block`; + got `64/8=8` but expected `24//2=12`). Overriding to `16` makes `16//2=8 == 8`. + Both the SFT launcher and the eval config set this. + +- **Hydra `+` prefix for `metadata_json_path`.** The key is *commented out* in the + config struct (not part of the Hydra schema), so it must be **added** with the + `+` prefix: `+actor.model.metadata_json_path=/fsx/models/metadata-libero.json`. + A plain override (without `+`) fails with "Key not in struct". Set + `METADATA_PATH=""` in the launcher to fall back to the checkpoint's bundled + `oxe_droid` metadata instead. + +- **DCP → `.pt` conversion is mandatory; do NOT save full model weights from + FSDP.** SFT writes a sharded DCP checkpoint; eval needs a single `.pt`. Convert + offline on CPU (step 5). Do **not** pass + `+actor.fsdp_config.save_full_model_weights=true` — the rank-0 full-state-dict + gather for the 16B model stalls / never completes on 2× `p5en.48xlarge`. + +- **14B eval config resolves the 14B architecture via Hydra `searchpath`.** The + embodiment eval ships only `model/dreamzero_5b` under its config dir; the 14B + arch (`model/dreamzero_14b.yaml`) lives in the SFT config tree. The eval config + adds `examples/sft/config/` to the Hydra `searchpath` so + `model/dreamzero_14b@actor.model` resolves. Component pretrained paths are left + `null` — the backbone comes from the DreamZero-DROID safetensors + your + `full_weights.pt`, not a 5B Wan download. + +## Troubleshooting + +| Symptom | Cause | Fix | +|---------|-------|-----| +| Pods crash with `ray: command not found`, or inline `${...}` vars render empty | Manifest rendered with unrestricted `envsubst`, clobbering inline shell vars | Always render manifests that embed shell with the **restricted** allow-list: `envsubst '${ECR_URI} ${NAMESPACE}' < ...`. | +| `ray: command not found` on Ray head/worker/submitter | KubeRay runs `ray` non-interactively (`~/.bashrc` not sourced); `ray` lives in the `dreamzero` venv | Already handled in `dreamzero-sft.yaml`: the venv `bin` is prepended to `PATH` on the head/worker `env` **and** on the `submitterPodTemplate`. Don't strip those `PATH` entries. | +| SFT crashes near checkpoint save: `UnpicklingError: invalid load key '\x00'` in `broadcast_object_list`, after shards are written | A best-effort DCP-save finalize step crashes *after* the on-disk shards are already complete | The in-image best-effort save patch handles this; the **on-disk checkpoint is valid and convertible**. Proceed to step 5 (convert). | +| Convert fails with `EOFError` / `inline_container.cc unexpected pos`, corrupt shards | FSx filled up during the SFT run; `torch.save` was truncated mid-write | Ensure **≥250 GB free** on FSx before SFT (a 14B DCP checkpoint is ~140–206 GB). Free space, re-run SFT. | +| SFT fails with `KeyError: embodiment_tag 'libero_sim' not found` | The DROID checkpoint only bundles `oxe_droid` metadata | Run step 3 (`generate-metadata.yaml`) to produce `/fsx/models/metadata-libero.json` before SFT/eval. | +| Metadata generation or transforms behave wrongly / schema errors | Wrong dataset (`lerobot/libero` has a different `observation.state`/`action` schema) | The dataset **must** be `physical-intelligence/libero`. Re-stage step 2. | +| Forward-pass assertion `actions … != … // …` early in SFT/eval | `num_action_per_block` inherited DROID default of 24 | Override `actor.model.num_action_per_block=16` (the launcher and eval config already do this). | +| Eval CUDA OOM at step 0 (GPU 0 ~280 MB free) | `total_num_envs=128` (upstream 5B default) OOMs the 16.48B model co-located with the sim | Use `total_num_envs=16` (the 14B config default). For a quick smoke eval, override to `8` via `HYDRA_OVERRIDES`. | +| `eval/success_once = 0.0` | The checkpoint came from a 1-step (validation) SFT run | **Expected** for a 1-step checkpoint. Run a multi-step SFT (step 4, raise `runner.max_steps`), re-convert, re-eval for real accuracy. | + +## Software versions + +| Component | Version | +|-----------|---------| +| RLinf (upstream) | `b3bbabb1f461` | +| DreamZero / `groot` | `ab790c198fbc` | +| EFA installer | 1.47.0 | +| libfabric | 2.4.0 | +| aws-ofi-nccl | 1.18.0 | +| NCCL | v2.21.5-1 | +| KubeRay / Ray | 1.6.0 / 2.55.1 | +| PyTorch | 2.6.0+cu124 | +| diffusers | 0.37.1 | +| lerobot | 0.3.3 | +| torchcodec | 0.2 | + +## References + +- RLinf training framework — [github.com/RLinf/RLinf](https://github.com/RLinf/RLinf) +- DreamZero (`groot`) model code — [github.com/RLinf/dreamzero](https://github.com/RLinf/dreamzero) +- DreamZero-DROID checkpoint — [huggingface.co/GEAR-Dreams/DreamZero-DROID](https://huggingface.co/GEAR-Dreams/DreamZero-DROID) +- UMT5-XXL tokenizer — [huggingface.co/google/umt5-xxl](https://huggingface.co/google/umt5-xxl) +- LIBERO dataset — [huggingface.co/datasets/physical-intelligence/libero](https://huggingface.co/datasets/physical-intelligence/libero) +- EKS cluster architectures — [`1.architectures/4.amazon-eks`](../../../../../1.architectures/4.amazon-eks) + +## Security + +See [CONTRIBUTING](https://github.com/aws-samples/awsome-distributed-training/blob/main/CONTRIBUTING.md#security-issue-notifications) +for more information. Credentials (Hugging Face tokens, etc.) flow through +Kubernetes Secrets referenced by `secretKeyRef`, never committed to rendered +YAML; `env_vars` is gitignored. + +## License + +This project is licensed under the MIT-0 License. See the LICENSE file. From e82f83162f113a6f9b4f08a6701e431075349be9 Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Wed, 17 Jun 2026 12:22:57 -0500 Subject: [PATCH 10/45] docs(dreamzero): add top-level test-case README --- 3.test_cases/pytorch/dreamzero/README.md | 81 ++++++++++++++++++++++++ 1 file changed, 81 insertions(+) create mode 100644 3.test_cases/pytorch/dreamzero/README.md diff --git a/3.test_cases/pytorch/dreamzero/README.md b/3.test_cases/pytorch/dreamzero/README.md new file mode 100644 index 000000000..415e1f220 --- /dev/null +++ b/3.test_cases/pytorch/dreamzero/README.md @@ -0,0 +1,81 @@ + + + +# DreamZero LIBERO 14B SFT (World-Action Model) on Amazon EKS + +DreamZero is a **16.48B-parameter World-Action Model (WAM)** — a Wan-based +video-diffusion Diffusion Transformer that *jointly* denoises future video frames +and robot actions in a shared causal self-attention space. This test case +continues supervised fine-tuning from the released +[`GEAR-Dreams/DreamZero-DROID`](https://huggingface.co/GEAR-Dreams/DreamZero-DROID) +14B checkpoint onto a *new* embodiment (LIBERO) and evaluates it in the LIBERO +simulator on Amazon EKS — demonstrating cross-embodiment transfer and multi-node +**FSDP2** training over **EFA** with **KubeRay**. Upstream: +[github.com/RLinf/RLinf](https://github.com/RLinf/RLinf) (training framework) and +[github.com/RLinf/dreamzero](https://github.com/RLinf/dreamzero) (the `groot` WAM +model code). + +## Architecture + +![DreamZero WAM](diagrams/dreamzero-wam.drawio.svg) + +## Layout + +``` +dreamzero/ +├── Dockerfile # two-stage RLinf + EFA overlay image +├── docker/ # build-context helpers (install_extras, DCP-save patch) +├── diagrams/ # WAM + infra topology (draw.io + SVG) +├── assets/ # rollout video + loss curve +└── kubernetes/libero/ # the EKS recipe (RayJob SFT + eval) — see its README +``` + +## Hardware + +**2× `p5en.48xlarge`** (8× NVIDIA H200 each = **16 GPUs total**), **16 EFA NICs per +node**, and **FSx for Lustre with ≥250 GB free** (a 14B FSDP DCP checkpoint is +~140–206 GB). + +## Prerequisites + +An Amazon EKS cluster with GPU autoscaling (Karpenter), EFA networking, the +KubeRay operator, and FSx for Lustre shared storage. See +[`../../1.architectures/4.amazon-eks`](../../1.architectures/4.amazon-eks) for +cluster setup. The detailed prerequisite checklist lives in the walkthrough below. + +## Full walkthrough + +**Full step-by-step walkthrough: [`kubernetes/libero/README.md`](kubernetes/libero/README.md)** — +build-image → push-to-ECR → stage models/dataset → generate metadata → multi-node +SFT RayJob → DCP→`.pt` conversion → LIBERO simulator eval. + +## Results / validation status + +The pipeline was validated **end-to-end** with a **1-step SFT smoke run** on 2× +`p5en.48xlarge`: the KubeRay RayJob reached `SUCCEEDED`, a **209 GB sharded FSDP +DCP checkpoint** was written, and it was converted to a single **91.7 GB `.pt`** +on CPU and consumed by the LIBERO simulator eval. This proves the *infrastructure +and pipeline* (image build, multi-node EFA/NCCL, FSDP2 sharded checkpointing, +DCP→`.pt` conversion, and in-sim eval) — **not** task accuracy. A 1-step +checkpoint yields `eval/success_once = 0.0`, which is **expected**; real accuracy +requires a multi-step SFT run (raise `runner.max_steps`). + +The **local `docker buildx` build path is the validated primary**; the in-cluster +kaniko alternative is provided but **pending live validation**. + +## References + +- RLinf training framework — [github.com/RLinf/RLinf](https://github.com/RLinf/RLinf) +- DreamZero (`groot`) model code — [github.com/RLinf/dreamzero](https://github.com/RLinf/dreamzero) +- DreamZero-DROID checkpoint — [huggingface.co/GEAR-Dreams/DreamZero-DROID](https://huggingface.co/GEAR-Dreams/DreamZero-DROID) +- EKS cluster architectures — [`1.architectures/4.amazon-eks`](../../1.architectures/4.amazon-eks) + +## Security + +See [CONTRIBUTING](https://github.com/aws-samples/awsome-distributed-training/blob/main/CONTRIBUTING.md#security-issue-notifications) +for more information. Credentials (Hugging Face tokens, etc.) flow through +Kubernetes Secrets, never committed to rendered YAML. + +## License + +This project is licensed under the MIT-0 License. See the LICENSE file. From e73e464c2c60d4051c291dbe6daa06d7458b605b Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Wed, 17 Jun 2026 12:24:42 -0500 Subject: [PATCH 11/45] chore(dreamzero): add MIT-0 headers to ported scripts/config; de-couple KubeRay prereq comment Final self-contained sweep: add SPDX MIT-0 headers to the 5 ported shell scripts + the eval config (carried over from RLinf-on-eks without headers); change the SFT manifest's KubeRay-prereq comment from the RLinf-on-eks terraform reference to the generic helm install. Repo is now FULLY_SELF_CONTAINED. --- 3.test_cases/pytorch/dreamzero/docker/scripts/install_extras.sh | 2 ++ .../pytorch/dreamzero/docker/scripts/run_training_eks.sh | 2 ++ .../pytorch/dreamzero/kubernetes/libero/dreamzero-sft.yaml | 2 +- .../dreamzero/kubernetes/libero/scripts/convert_checkpoint.sh | 2 ++ .../libero/scripts/libero_spatial_eval_dreamzero_14b.yaml | 2 ++ .../kubernetes/libero/scripts/run_dreamzero_eval_eks.sh | 2 ++ .../kubernetes/libero/scripts/run_dreamzero_sft_eks.sh | 2 ++ 7 files changed, 13 insertions(+), 1 deletion(-) diff --git a/3.test_cases/pytorch/dreamzero/docker/scripts/install_extras.sh b/3.test_cases/pytorch/dreamzero/docker/scripts/install_extras.sh index c4bc12edb..9a70bc5e2 100755 --- a/3.test_cases/pytorch/dreamzero/docker/scripts/install_extras.sh +++ b/3.test_cases/pytorch/dreamzero/docker/scripts/install_extras.sh @@ -1,4 +1,6 @@ #!/bin/bash +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 # ============================================================================= # Install optional extras for RLinf-on-EKS container image # diff --git a/3.test_cases/pytorch/dreamzero/docker/scripts/run_training_eks.sh b/3.test_cases/pytorch/dreamzero/docker/scripts/run_training_eks.sh index 2df8db654..325ef4423 100644 --- a/3.test_cases/pytorch/dreamzero/docker/scripts/run_training_eks.sh +++ b/3.test_cases/pytorch/dreamzero/docker/scripts/run_training_eks.sh @@ -1,4 +1,6 @@ #!/bin/bash +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 # ============================================================================= # RLinf Training Launch Script for Amazon EKS # diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-sft.yaml b/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-sft.yaml index 1e58484f9..a2f34e4a2 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-sft.yaml +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-sft.yaml @@ -16,7 +16,7 @@ # - shutdownAfterJobFinishes tears the RayCluster down when the job completes. # # Prerequisites: -# - KubeRay operator installed (infrastructure/addons, enable_kuberay=true) +# - KubeRay operator installed (helm install kuberay-operator, chart 1.6.0) # - dataset/model staging job completed # (DreamZero-DROID warm-start + umt5-xxl tokenizer + LIBERO dataset on FSx) # - FSx PVC "fsx-claim" bound, >=250GB free diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/convert_checkpoint.sh b/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/convert_checkpoint.sh index 9d1b9d7f8..52b585c32 100755 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/convert_checkpoint.sh +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/convert_checkpoint.sh @@ -1,4 +1,6 @@ #!/bin/bash +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 # ============================================================================= # DreamZero checkpoint conversion: FSDP DCP shards -> single .pt # diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/libero_spatial_eval_dreamzero_14b.yaml b/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/libero_spatial_eval_dreamzero_14b.yaml index ea54019a4..84fadf62a 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/libero_spatial_eval_dreamzero_14b.yaml +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/libero_spatial_eval_dreamzero_14b.yaml @@ -1,3 +1,5 @@ +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 # libero_spatial_eval_dreamzero_14b.yaml # ============================================================================= # DreamZero 14B LIBERO-Spatial simulator eval config (RLinf-on-EKS variant). diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/run_dreamzero_eval_eks.sh b/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/run_dreamzero_eval_eks.sh index 9c85779ae..052fc9cf2 100755 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/run_dreamzero_eval_eks.sh +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/run_dreamzero_eval_eks.sh @@ -1,4 +1,6 @@ #!/bin/bash +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 # ============================================================================= # DreamZero LIBERO simulator eval launcher (single-node, GPU). # diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/run_dreamzero_sft_eks.sh b/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/run_dreamzero_sft_eks.sh index 1a466845a..a39804203 100755 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/run_dreamzero_sft_eks.sh +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/run_dreamzero_sft_eks.sh @@ -1,4 +1,6 @@ #!/bin/bash +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 # ============================================================================= # DreamZero LIBERO SFT Launch Script for Amazon EKS (Ray / FSDP2) # From f6c80234eb0dda94427f1f5ad6be738bcfff0132 Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Wed, 17 Jun 2026 16:19:43 -0500 Subject: [PATCH 12/45] fix(dreamzero): correct infra diagram to KubeRay RayJob; drop stale assets The infra-dreamzero-sft topology diagram depicted the abandoned StatefulSet + headless Service design (manual `ray start` head election, CodeBuild build step, leaked namespace), contradicting both the RayJob manifest and the README prose beside it. Rewrite it to the validated KubeRay RayJob topology: operator-managed embedded RayCluster (1 head + 1 worker), no manual head election, tool-neutral "build + push to ECR" step, placeholder, and a bidirectional RayJob<->FSx arrow showing the /fsx mount (reads model/dataset/metadata, writes DCP checkpoints). Re-export SVG (white/dark adaptive background to match the sibling diagrams). Remove the rollout video and loss curve: the 1-step smoke-run rollout (success_once=0, expected) reads as broken in a public PR, and the loss curve is from a pre-refactor DeepSpeed run now disconnected from the FSDP2 config. README text describes the validated scope instead. Drop the now-unused assets/*.mp4 and assets/*.png LFS attributes. --- 3.test_cases/pytorch/dreamzero/.gitattributes | 2 - 3.test_cases/pytorch/dreamzero/README.md | 1 - .../dreamzero/assets/libero-rollout-seed0.mp4 | 3 - .../pytorch/dreamzero/assets/loss-curve.png | 3 - .../diagrams/infra-dreamzero-sft.drawio | 200 ++++++++++-------- .../diagrams/infra-dreamzero-sft.drawio.svg | 4 +- 6 files changed, 116 insertions(+), 97 deletions(-) delete mode 100644 3.test_cases/pytorch/dreamzero/assets/libero-rollout-seed0.mp4 delete mode 100644 3.test_cases/pytorch/dreamzero/assets/loss-curve.png diff --git a/3.test_cases/pytorch/dreamzero/.gitattributes b/3.test_cases/pytorch/dreamzero/.gitattributes index f416907f4..4f973d2f8 100644 --- a/3.test_cases/pytorch/dreamzero/.gitattributes +++ b/3.test_cases/pytorch/dreamzero/.gitattributes @@ -1,3 +1 @@ -assets/*.mp4 filter=lfs diff=lfs merge=lfs -text -assets/*.png filter=lfs diff=lfs merge=lfs -text diagrams/*.svg filter=lfs diff=lfs merge=lfs -text diff --git a/3.test_cases/pytorch/dreamzero/README.md b/3.test_cases/pytorch/dreamzero/README.md index 415e1f220..edb2fc1eb 100644 --- a/3.test_cases/pytorch/dreamzero/README.md +++ b/3.test_cases/pytorch/dreamzero/README.md @@ -26,7 +26,6 @@ dreamzero/ ├── Dockerfile # two-stage RLinf + EFA overlay image ├── docker/ # build-context helpers (install_extras, DCP-save patch) ├── diagrams/ # WAM + infra topology (draw.io + SVG) -├── assets/ # rollout video + loss curve └── kubernetes/libero/ # the EKS recipe (RayJob SFT + eval) — see its README ``` diff --git a/3.test_cases/pytorch/dreamzero/assets/libero-rollout-seed0.mp4 b/3.test_cases/pytorch/dreamzero/assets/libero-rollout-seed0.mp4 deleted file mode 100644 index 81dcfa537..000000000 --- a/3.test_cases/pytorch/dreamzero/assets/libero-rollout-seed0.mp4 +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:b8a735fe4ac0185deb56ed54054e414a165376d042b662fe3fd283cb058630cc -size 61217 diff --git a/3.test_cases/pytorch/dreamzero/assets/loss-curve.png b/3.test_cases/pytorch/dreamzero/assets/loss-curve.png deleted file mode 100644 index 8e46c78df..000000000 --- a/3.test_cases/pytorch/dreamzero/assets/loss-curve.png +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:6c99d97ce8e928af261fe2cd1ae9dba45c3e9d978062fbc57d76b21effe039a2 -size 147648 diff --git a/3.test_cases/pytorch/dreamzero/diagrams/infra-dreamzero-sft.drawio b/3.test_cases/pytorch/dreamzero/diagrams/infra-dreamzero-sft.drawio index 10cb08540..a300c51d5 100644 --- a/3.test_cases/pytorch/dreamzero/diagrams/infra-dreamzero-sft.drawio +++ b/3.test_cases/pytorch/dreamzero/diagrams/infra-dreamzero-sft.drawio @@ -1,86 +1,114 @@ - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/3.test_cases/pytorch/dreamzero/diagrams/infra-dreamzero-sft.drawio.svg b/3.test_cases/pytorch/dreamzero/diagrams/infra-dreamzero-sft.drawio.svg index a1686b168..c953055c4 100644 --- a/3.test_cases/pytorch/dreamzero/diagrams/infra-dreamzero-sft.drawio.svg +++ b/3.test_cases/pytorch/dreamzero/diagrams/infra-dreamzero-sft.drawio.svg @@ -1,3 +1,3 @@ version https://git-lfs.github.com/spec/v1 -oid sha256:b17ad4438637ffb367e61c975739b313ecd03f47950ed9f97c443e821a8a4c3b -size 761669 +oid sha256:51336b6d65219fcffc168de913a9ba134a3fa742d71c8ad419f4ae96318db5f1 +size 986509 From b7a17332da87c48ca85f72396eb8bd399ac6e51c Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Thu, 18 Jun 2026 00:07:14 -0500 Subject: [PATCH 13/45] fix(dreamzero): root-cause DCP finalization crash (gloo coordinator PG) + force save_full_model_weights=false Reproduced, root-caused, and fixed the DreamZero 16B SFT checkpoint crash end-to-end on 2x p5en (RayJob SUCCEEDED, 207GB DCP, zero errors). Root cause: on torch 2.6, dcp.save's post-write finalization broadcasts a multi-MB pickled result object over the default (NCCL) process group on CUDA. At the end of a long (~209GB / ~20min) checkpoint write this races with NCCL comm teardown, leaving non-coordinator ranks with an all-zero buffer -> `_pickle.UnpicklingError: invalid load key '\x00'`, AFTER all 16 shards + .metadata are already on disk. Fix (dcp-save-gloo-coordinator.patch): pass a dedicated CPU/gloo process group to dcp.save(..., process_group=gloo_pg) so the finalization object-broadcast runs over gloo (CPU), immune to the CUDA/NCCL teardown race -- the same approach torch 2.7+ takes upstream. Replaces the prior symptom-guard dcp-save-finalize-besteffort.patch (removed). Also force +actor.fsdp_config.save_full_model_weights=false in the launcher: libero_sft_dreamzero_14b.yaml omits the key, so it defaults to True, which on the 16B model hits "Backend nccl does not support allgather_into_tensor_coalesced" during the full-state-dict gather. DCP-only + offline convert is the supported path. Harden the Dockerfile patch-apply loop to be nullglob-safe. Update the kubernetes/libero README + kaniko-build.yaml patch references accordingly. --- 3.test_cases/pytorch/dreamzero/Dockerfile | 12 +++-- .../dcp-save-finalize-besteffort.patch | 46 ------------------- .../patches/dcp-save-gloo-coordinator.patch | 38 +++++++++++++++ .../dreamzero/kubernetes/libero/README.md | 6 +-- .../libero/scripts/run_dreamzero_sft_eks.sh | 10 ++++ .../kubernetes/libero/setup/kaniko-build.yaml | 6 +-- 6 files changed, 62 insertions(+), 56 deletions(-) delete mode 100644 3.test_cases/pytorch/dreamzero/docker/scripts/patches/dcp-save-finalize-besteffort.patch create mode 100644 3.test_cases/pytorch/dreamzero/docker/scripts/patches/dcp-save-gloo-coordinator.patch diff --git a/3.test_cases/pytorch/dreamzero/Dockerfile b/3.test_cases/pytorch/dreamzero/Dockerfile index 4134d86f2..8bc4cf368 100644 --- a/3.test_cases/pytorch/dreamzero/Dockerfile +++ b/3.test_cases/pytorch/dreamzero/Dockerfile @@ -209,14 +209,18 @@ COPY RLinf/ /workspace/RLinf/ COPY DreamZero/ /workspace/DreamZero/ # Apply EKS hotfix patches to the upstream RLinf source. -# dcp-save-finalize-besteffort.patch: tolerate torch DCP's post-write -# finalization broadcast crash (UnpicklingError on multi-node gloo PG) when the -# on-disk checkpoint is already complete. Must be baked into the image because +# dcp-save-gloo-coordinator.patch: pass a CPU/gloo process group to dcp.save so +# its post-write finalization object-broadcast runs over gloo instead of CUDA/NCCL. +# On torch 2.6 the default-PG (NCCL) broadcast of the multi-MB finalization object +# races with NCCL comm teardown at the end of a long (100s+ GB) checkpoint write, +# leaving non-coordinator ranks with an all-zero buffer -> UnpicklingError, AFTER +# every shard + .metadata are already on disk (reproduced + validated on 2x p5en; +# torch 2.7+ fixes this upstream the same way). Must be baked into the image because # Ray worker actors on every node import from /workspace/RLinf. git is present # (installed in the EFA stage). Patches are validated to apply against the pinned # UPSTREAM_REF; the build fails loudly if a patch no longer applies. RUN cd /workspace/RLinf && \ - for p in /workspace/eks/scripts/patches/*.patch; do \ + for p in /workspace/eks/scripts/patches/*.patch; do [ -e "$p" ] || continue; \ echo "Applying patch: ${p}"; \ git apply --verbose "${p}" || patch -p1 < "${p}"; \ done diff --git a/3.test_cases/pytorch/dreamzero/docker/scripts/patches/dcp-save-finalize-besteffort.patch b/3.test_cases/pytorch/dreamzero/docker/scripts/patches/dcp-save-finalize-besteffort.patch deleted file mode 100644 index 6b8c9a744..000000000 --- a/3.test_cases/pytorch/dreamzero/docker/scripts/patches/dcp-save-finalize-besteffort.patch +++ /dev/null @@ -1,46 +0,0 @@ -diff --git a/rlinf/hybrid_engines/fsdp/strategy/base.py b/rlinf/hybrid_engines/fsdp/strategy/base.py -index 06bdd2d4..3f757111 100644 ---- a/rlinf/hybrid_engines/fsdp/strategy/base.py -+++ b/rlinf/hybrid_engines/fsdp/strategy/base.py -@@ -233,10 +233,37 @@ class FSDPStrategyBase(ABC): - from torch.distributed import checkpoint as dcp - - dcp_save_path = os.path.join(save_path, "dcp_checkpoint") -- dcp.save( -- {"fsdp_checkpoint": training_state}, -- checkpoint_id=dcp_save_path, -- ) -+ try: -+ dcp.save( -+ {"fsdp_checkpoint": training_state}, -+ checkpoint_id=dcp_save_path, -+ ) -+ except BaseException as dcp_err: -+ # EKS hotfix: torch DCP's post-write finalization broadcast -+ # (all_reduce("write", ...) -> broadcast_object_list) can raise -+ # `_pickle.UnpicklingError: invalid load key, '\x00'` on multi-node -+ # gloo process groups, AFTER write_data() and finish_checkpoint() -+ # have already written every shard + .metadata to disk. Treat the -+ # save as successful when the on-disk DCP is complete; otherwise -+ # re-raise (genuine write failure, e.g. truncated/full filesystem). -+ import glob as _glob -+ -+ meta_ok = os.path.isfile( -+ os.path.join(dcp_save_path, ".metadata") -+ ) -+ shards = _glob.glob(os.path.join(dcp_save_path, "*.distcp")) -+ world = torch.distributed.get_world_size() -+ if meta_ok and len(shards) >= world: -+ if hasattr(cls, "logger") and cls.logger is not None: -+ cls.logger.warning( -+ "dcp.save raised in post-write finalization " -+ f"({type(dcp_err).__name__}: {dcp_err}) but the " -+ f"on-disk checkpoint is complete (.metadata + " -+ f"{len(shards)} shards >= world_size {world}) at " -+ f"{dcp_save_path}; treating save as successful." -+ ) -+ else: -+ raise - - except BaseException as e: - import traceback diff --git a/3.test_cases/pytorch/dreamzero/docker/scripts/patches/dcp-save-gloo-coordinator.patch b/3.test_cases/pytorch/dreamzero/docker/scripts/patches/dcp-save-gloo-coordinator.patch new file mode 100644 index 000000000..fdbecf528 --- /dev/null +++ b/3.test_cases/pytorch/dreamzero/docker/scripts/patches/dcp-save-gloo-coordinator.patch @@ -0,0 +1,38 @@ +--- a/rlinf/hybrid_engines/fsdp/strategy/base.py ++++ b/rlinf/hybrid_engines/fsdp/strategy/base.py +@@ -233,9 +233,35 @@ + from torch.distributed import checkpoint as dcp + + dcp_save_path = os.path.join(save_path, "dcp_checkpoint") ++ # EKS root-cause fix (torch 2.6 DCP): dcp.save's post-write ++ # finalization runs all_reduce("write", write_data, ++ # finish_checkpoint), whose broadcast_object step pickles a ++ # multi-MB result object and broadcasts it over the process ++ # group. With the default NCCL/CUDA group, that object broadcast ++ # runs on CUDA and races with NCCL comm teardown at the end of a ++ # long (100s+ GB) checkpoint write -> non-coordinator ranks read ++ # an all-zero buffer -> `_pickle.UnpicklingError: invalid load ++ # key '\x00'`, AFTER every shard + .metadata are already on disk. ++ # torch 2.7+ fixes this by running DCP coordination on a CPU/gloo ++ # PG; backport that by passing a dedicated gloo process group to ++ # dcp.save so the object broadcast goes over CPU/gloo (immune to ++ # the CUDA/NCCL teardown race). ++ import datetime as _dt ++ ++ gloo_pg = getattr(cls, "_dcp_gloo_pg", None) ++ if gloo_pg is None and torch.distributed.is_initialized(): ++ try: ++ gloo_pg = torch.distributed.new_group( ++ backend="gloo", ++ timeout=_dt.timedelta(minutes=30), ++ ) ++ cls._dcp_gloo_pg = gloo_pg ++ except BaseException: ++ gloo_pg = None + dcp.save( + {"fsdp_checkpoint": training_state}, + checkpoint_id=dcp_save_path, ++ process_group=gloo_pg, + ) + + except BaseException as e: diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md index 65fdd90b5..eeed0a3f4 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md @@ -269,7 +269,7 @@ not clobbered to empty strings. The image is built in two stages: the upstream RLinf `embodied-libero` target (which natively builds the dedicated **`dreamzero` venv**, RLinf PR #1272), then an EFA overlay (`Dockerfile` at the test-case root) that layers EFA 1.47.0 / -libfabric 2.4.0 / aws-ofi-nccl 1.18.0 and a best-effort DCP-save patch. The +libfabric 2.4.0 / aws-ofi-nccl 1.18.0 and the DCP-save gloo-coordinator patch. The external `groot` package is cloned from `github.com/RLinf/dreamzero.git` and placed on `PYTHONPATH` via `DREAMZERO_PATH=/workspace/DreamZero`. @@ -306,7 +306,7 @@ stage 1 to complete before stage 2): kubectl -n "$NAMESPACE" create configmap dreamzero-build-context \ --from-file=Dockerfile=Dockerfile \ --from-file=install_extras.sh=docker/scripts/install_extras.sh \ - --from-file=dcp-save-finalize-besteffort.patch=docker/scripts/patches/dcp-save-finalize-besteffort.patch \ + --from-file=dcp-save-gloo-coordinator.patch=docker/scripts/patches/dcp-save-gloo-coordinator.patch \ --dry-run=client -o yaml | kubectl apply -f - # Apply both stages (restricted envsubst), then wait on stage 1 before stage 2 runs: @@ -497,7 +497,7 @@ config: |---------|-------|-----| | Pods crash with `ray: command not found`, or inline `${...}` vars render empty | Manifest rendered with unrestricted `envsubst`, clobbering inline shell vars | Always render manifests that embed shell with the **restricted** allow-list: `envsubst '${ECR_URI} ${NAMESPACE}' < ...`. | | `ray: command not found` on Ray head/worker/submitter | KubeRay runs `ray` non-interactively (`~/.bashrc` not sourced); `ray` lives in the `dreamzero` venv | Already handled in `dreamzero-sft.yaml`: the venv `bin` is prepended to `PATH` on the head/worker `env` **and** on the `submitterPodTemplate`. Don't strip those `PATH` entries. | -| SFT crashes near checkpoint save: `UnpicklingError: invalid load key '\x00'` in `broadcast_object_list`, after shards are written | A best-effort DCP-save finalize step crashes *after* the on-disk shards are already complete | The in-image best-effort save patch handles this; the **on-disk checkpoint is valid and convertible**. Proceed to step 5 (convert). | +| SFT crashes near checkpoint save: `UnpicklingError: invalid load key '\x00'` in `broadcast_object_list`, after shards are written | torch 2.6 DCP broadcasts its finalization object over the default NCCL/CUDA PG, which races with NCCL teardown at the end of a long write | Fixed by the in-image `dcp-save-gloo-coordinator.patch` (routes the broadcast over a CPU/gloo PG, like torch 2.7+). If you somehow hit this, the **on-disk checkpoint is still valid and convertible** — proceed to step 5 (convert). | | Convert fails with `EOFError` / `inline_container.cc unexpected pos`, corrupt shards | FSx filled up during the SFT run; `torch.save` was truncated mid-write | Ensure **≥250 GB free** on FSx before SFT (a 14B DCP checkpoint is ~140–206 GB). Free space, re-run SFT. | | SFT fails with `KeyError: embodiment_tag 'libero_sim' not found` | The DROID checkpoint only bundles `oxe_droid` metadata | Run step 3 (`generate-metadata.yaml`) to produce `/fsx/models/metadata-libero.json` before SFT/eval. | | Metadata generation or transforms behave wrongly / schema errors | Wrong dataset (`lerobot/libero` has a different `observation.state`/`action` schema) | The dataset **must** be `physical-intelligence/libero`. Re-stage step 2. | diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/run_dreamzero_sft_eks.sh b/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/run_dreamzero_sft_eks.sh index a39804203..36f649316 100755 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/run_dreamzero_sft_eks.sh +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/run_dreamzero_sft_eks.sh @@ -121,6 +121,16 @@ HYDRA_ARGS="${HYDRA_ARGS} runner.logger.log_path=${LOG_DIR}" # (got 64/8=8, expected 24//2=12). With num_action_per_block=16: 16//2=8 == 8. HYDRA_ARGS="${HYDRA_ARGS} actor.model.num_action_per_block=16" +# Force DCP-only checkpointing (no full-weights gather). libero_sft_dreamzero_14b.yaml +# omits fsdp_config.save_full_model_weights, so it falls through to the code default +# of True (rlinf/hybrid_engines/fsdp/fsdp_model_manager.py). On the 16B model that +# triggers a full-state-dict gather -> "Backend nccl does not support +# allgather_into_tensor_coalesced" and (per upstream) a rank-0 gather that stalls. +# The sharded DCP checkpoint is the supported path; convert it to a single .pt +# offline on CPU (convert-checkpoint.yaml). The key is absent from the config struct, +# so it must be ADDED with the '+' prefix. +HYDRA_ARGS="${HYDRA_ARGS} +actor.fsdp_config.save_full_model_weights=false" + # Metadata handling: set metadata_json_path when METADATA_PATH is non-empty. # It now DEFAULTS to /fsx/models/metadata-libero.json (LIBERO libero_sim stats # generated by generate-metadata.yaml), so the override is passed by default. diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/setup/kaniko-build.yaml b/3.test_cases/pytorch/dreamzero/kubernetes/libero/setup/kaniko-build.yaml index 4674e0862..f5b0be113 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/setup/kaniko-build.yaml +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/setup/kaniko-build.yaml @@ -8,7 +8,7 @@ # stage 1: build the upstream RLinf embodied-libero image, push as # ${ECR_URI}:upstream-embodied-libero # stage 2: FROM that tag, layer the EFA overlay (this test case's Dockerfile), -# apply the DCP-save patch, push ${ECR_URI}:latest +# apply the DCP-save gloo-coordinator patch, push ${ECR_URI}:latest # # A shared FSx scratch dir (PVC fsx-claim) holds the cloned sources + build cache. # The test-case Dockerfile + docker/ build helpers are delivered via a ConfigMap @@ -18,7 +18,7 @@ # kubectl -n ${NAMESPACE} create configmap dreamzero-build-context \ # --from-file=Dockerfile=Dockerfile \ # --from-file=install_extras.sh=docker/scripts/install_extras.sh \ -# --from-file=dcp-save-finalize-besteffort.patch=docker/scripts/patches/dcp-save-finalize-besteffort.patch \ +# --from-file=dcp-save-gloo-coordinator.patch=docker/scripts/patches/dcp-save-gloo-coordinator.patch \ # --dry-run=client -o yaml | kubectl apply -f - # envsubst '${ECR_URI} ${NAMESPACE}' < kubernetes/libero/setup/kaniko-build.yaml | kubectl apply -f - --- @@ -102,7 +102,7 @@ spec: mkdir -p /ctx/overlay/docker/scripts/patches cp /cm/Dockerfile /ctx/overlay/Dockerfile cp /cm/install_extras.sh /ctx/overlay/docker/scripts/install_extras.sh - cp /cm/dcp-save-finalize-besteffort.patch /ctx/overlay/docker/scripts/patches/dcp-save-finalize-besteffort.patch + cp /cm/dcp-save-gloo-coordinator.patch /ctx/overlay/docker/scripts/patches/dcp-save-gloo-coordinator.patch git clone https://github.com/RLinf/RLinf.git /ctx/overlay/RLinf git -C /ctx/overlay/RLinf checkout b3bbabb1f461 git clone https://github.com/RLinf/dreamzero.git /ctx/overlay/DreamZero From dbb6404743a3c43fdcf37acb5465a638f7194bb5 Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Thu, 18 Jun 2026 09:16:28 -0500 Subject: [PATCH 14/45] chore(dreamzero): drop unvalidated in-cluster kaniko build path Remove the experimental kaniko in-cluster build (setup/kaniko-build.yaml) and all references to it. Every other test case in the repo builds images with `docker build`/`docker buildx` and pushes to ECR; the closest analog (openvla-oft LIBERO-on-EKS) uses `docker buildx build --platform linux/amd64`. dreamzero was the only test case introducing kaniko, and that path was unvalidated, carried known build bugs, and pinned a `:latest` executor image (against CONTRIBUTING's "do not use a latest tag" rule). The validated `docker buildx` path (setup/build-push.sh) is now the sole, documented build method. kaniko can return in a follow-up PR once live-validated. - delete kubernetes/libero/setup/kaniko-build.yaml - README.md: build statement now points at build-push.sh - kubernetes/libero/README.md: remove the "Alternative path (kaniko)" block and the kaniko layout entry; reword the primary path as the sole path - Dockerfile: reword RLINF_UPSTREAM_IMAGE comment (kept the ARG; it is a generic stage-1 override the buildx path also uses) --- 3.test_cases/pytorch/dreamzero/Dockerfile | 6 +- 3.test_cases/pytorch/dreamzero/README.md | 4 +- .../dreamzero/kubernetes/libero/README.md | 34 +---- .../kubernetes/libero/setup/kaniko-build.yaml | 133 ------------------ 4 files changed, 8 insertions(+), 169 deletions(-) delete mode 100644 3.test_cases/pytorch/dreamzero/kubernetes/libero/setup/kaniko-build.yaml diff --git a/3.test_cases/pytorch/dreamzero/Dockerfile b/3.test_cases/pytorch/dreamzero/Dockerfile index 8bc4cf368..3098d7fdd 100644 --- a/3.test_cases/pytorch/dreamzero/Dockerfile +++ b/3.test_cases/pytorch/dreamzero/Dockerfile @@ -44,9 +44,9 @@ FROM rlinf-upstream-${BUILD_TARGET} AS upstream # Layer EFA, GDRCopy, NCCL, and OpenMPI onto the upstream RLinf image. # Recipe proven in nccl-tests/Dockerfile on nvidia/cuda base images. -# RLINF_UPSTREAM_IMAGE lets the kaniko build path (setup/kaniko-build.yaml) point -# stage 2 at the stage-1 image it pushed to ECR. The docker buildx path -# (setup/build-push.sh) uses the default local tag built immediately before. +# RLINF_UPSTREAM_IMAGE points stage 2 at the stage-1 image. The docker buildx +# path (setup/build-push.sh) uses the default local tag built immediately before; +# override it to pull a stage-1 image from a registry instead. ARG RLINF_UPSTREAM_IMAGE=rlinf-upstream-${BUILD_TARGET} FROM ${RLINF_UPSTREAM_IMAGE} diff --git a/3.test_cases/pytorch/dreamzero/README.md b/3.test_cases/pytorch/dreamzero/README.md index edb2fc1eb..b8994f80a 100644 --- a/3.test_cases/pytorch/dreamzero/README.md +++ b/3.test_cases/pytorch/dreamzero/README.md @@ -59,8 +59,8 @@ DCP→`.pt` conversion, and in-sim eval) — **not** task accuracy. A 1-step checkpoint yields `eval/success_once = 0.0`, which is **expected**; real accuracy requires a multi-step SFT run (raise `runner.max_steps`). -The **local `docker buildx` build path is the validated primary**; the in-cluster -kaniko alternative is provided but **pending live validation**. +The image is built with the **local `docker buildx`** two-stage build and pushed +to ECR (see [`kubernetes/libero/setup/build-push.sh`](kubernetes/libero/setup/build-push.sh)). ## References diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md index eeed0a3f4..d2b704ad0 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md @@ -273,8 +273,8 @@ libfabric 2.4.0 / aws-ofi-nccl 1.18.0 and the DCP-save gloo-coordinator patch. T external `groot` package is cloned from `github.com/RLinf/dreamzero.git` and placed on `PYTHONPATH` via `DREAMZERO_PATH=/workspace/DreamZero`. -**Primary path (local Docker + buildx).** Run from the `setup/` directory; it -reads `ECR_URI`, `AWS_REGION`, `UPSTREAM_REF`, and `DREAMZERO_REF` from +Build the image with **local Docker + buildx**. Run from the `setup/` directory; +it reads `ECR_URI`, `AWS_REGION`, `UPSTREAM_REF`, and `DREAMZERO_REF` from `env_vars`: ```bash @@ -288,33 +288,6 @@ This clones the pinned `RLinf` and `dreamzero` sources, builds stage 1 (`rlinf-upstream-embodied-libero`), then builds and pushes stage 2 to `${ECR_URI}:latest`. -**Alternative path (in-cluster kaniko, disk-constrained hosts).** - -> ⚠️ **EXPERIMENTAL — PENDING LIVE VALIDATION.** The kaniko in-cluster build path -> below is provided for hosts without a local Docker daemon or sufficient disk. -> It is **not-yet-validated** / **untested in-cluster as of this writing**. The -> primary `build-push.sh` path above is the validated one. Use kaniko at your own -> risk and expect to debug the two-stage build. - -The kaniko path (`setup/kaniko-build.yaml`) runs two Jobs against a shared FSx -scratch dir. First create the build-context ConfigMap from the test-case -`Dockerfile` and `docker/scripts/`, then apply the two stages in order (wait for -stage 1 to complete before stage 2): - -```bash -# From the test-case root (3.test_cases/pytorch/dreamzero), create the context CM: -kubectl -n "$NAMESPACE" create configmap dreamzero-build-context \ - --from-file=Dockerfile=Dockerfile \ - --from-file=install_extras.sh=docker/scripts/install_extras.sh \ - --from-file=dcp-save-gloo-coordinator.patch=docker/scripts/patches/dcp-save-gloo-coordinator.patch \ - --dry-run=client -o yaml | kubectl apply -f - - -# Apply both stages (restricted envsubst), then wait on stage 1 before stage 2 runs: -envsubst '${ECR_URI} ${NAMESPACE}' < kubernetes/libero/setup/kaniko-build.yaml | kubectl apply -f - -kubectl wait --for=condition=complete job/dreamzero-kaniko-stage1 -n "$NAMESPACE" --timeout=90m -kubectl logs -f job/dreamzero-kaniko-stage2 -n "$NAMESPACE" -``` - ### 2. Stage models + dataset to FSx Downloads the DreamZero-DROID 14B warm-start checkpoint @@ -450,8 +423,7 @@ kubernetes/libero/ │ ├── run_dreamzero_eval_eks.sh # LIBERO simulator eval launcher │ └── libero_spatial_eval_dreamzero_14b.yaml # 14B eval config (upstream ships 5B only) ├── setup/ -│ ├── build-push.sh # Primary: local buildx two-stage build + push -│ └── kaniko-build.yaml # EXPERIMENTAL in-cluster kaniko build (pending validation) +│ └── build-push.sh # Local buildx two-stage build + push to ECR └── storage/ ├── pvc-fsx-lustre-dynamic.yaml # OPTIONAL: dynamically provision fsx-claim └── pv-fsx-lustre-static.yaml # OPTIONAL: bind fsx-claim to an existing FSx diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/setup/kaniko-build.yaml b/3.test_cases/pytorch/dreamzero/kubernetes/libero/setup/kaniko-build.yaml deleted file mode 100644 index f5b0be113..000000000 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/setup/kaniko-build.yaml +++ /dev/null @@ -1,133 +0,0 @@ -# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. -# SPDX-License-Identifier: MIT-0 -# -# In-cluster image build via the Chainguard kaniko fork (no local Docker daemon -# or disk needed). Alternative to setup/build-push.sh for disk-constrained users. -# -# Two stages, run as two Jobs (apply stage 1, wait for completion, then stage 2): -# stage 1: build the upstream RLinf embodied-libero image, push as -# ${ECR_URI}:upstream-embodied-libero -# stage 2: FROM that tag, layer the EFA overlay (this test case's Dockerfile), -# apply the DCP-save gloo-coordinator patch, push ${ECR_URI}:latest -# -# A shared FSx scratch dir (PVC fsx-claim) holds the cloned sources + build cache. -# The test-case Dockerfile + docker/ build helpers are delivered via a ConfigMap -# (created from the test-case dir; see kubernetes/libero/README.md). -# -# Apply (restricted envsubst): -# kubectl -n ${NAMESPACE} create configmap dreamzero-build-context \ -# --from-file=Dockerfile=Dockerfile \ -# --from-file=install_extras.sh=docker/scripts/install_extras.sh \ -# --from-file=dcp-save-gloo-coordinator.patch=docker/scripts/patches/dcp-save-gloo-coordinator.patch \ -# --dry-run=client -o yaml | kubectl apply -f - -# envsubst '${ECR_URI} ${NAMESPACE}' < kubernetes/libero/setup/kaniko-build.yaml | kubectl apply -f - ---- -apiVersion: batch/v1 -kind: Job -metadata: - name: dreamzero-kaniko-stage1 - namespace: ${NAMESPACE} - labels: - app: dreamzero-kaniko - stage: "1" -spec: - backoffLimit: 0 - template: - metadata: - labels: - app: dreamzero-kaniko - stage: "1" - spec: - restartPolicy: Never - serviceAccountName: training-sa - initContainers: - - name: clone-rlinf - image: alpine/git:2.45.2 - command: ["sh", "-c"] - args: - - | - set -eux - rm -rf /ctx/RLinf - git clone https://github.com/RLinf/RLinf.git /ctx/RLinf - git -C /ctx/RLinf checkout b3bbabb1f461 - volumeMounts: - - { name: ctx, mountPath: /ctx } - containers: - - name: kaniko - image: ghcr.io/chainguard-forks/kaniko/executor:latest - args: - - --dockerfile=/ctx/RLinf/docker/Dockerfile - - --context=dir:///ctx/RLinf - - --build-arg=BUILD_TARGET=embodied-libero - - --build-arg=NO_MIRROR=1 - - --destination=${ECR_URI}:upstream-embodied-libero - - --cache=true - - --cache-repo=${ECR_URI}/cache - volumeMounts: - - { name: ctx, mountPath: /ctx } - volumes: - - name: ctx - persistentVolumeClaim: - claimName: fsx-claim ---- -apiVersion: batch/v1 -kind: Job -metadata: - name: dreamzero-kaniko-stage2 - namespace: ${NAMESPACE} - labels: - app: dreamzero-kaniko - stage: "2" -spec: - backoffLimit: 0 - template: - metadata: - labels: - app: dreamzero-kaniko - stage: "2" - spec: - restartPolicy: Never - serviceAccountName: training-sa - initContainers: - - name: assemble-context - image: alpine/git:2.45.2 - command: ["sh", "-c"] - args: - - | - set -eux - # Stage-2 build context = /ctx/overlay containing the test-case - # Dockerfile + docker/scripts/ (helpers + DCP patch) + the DreamZero - # groot clone. RLinf source is re-cloned for the COPY RLinf/ layer. - rm -rf /ctx/overlay - mkdir -p /ctx/overlay/docker/scripts/patches - cp /cm/Dockerfile /ctx/overlay/Dockerfile - cp /cm/install_extras.sh /ctx/overlay/docker/scripts/install_extras.sh - cp /cm/dcp-save-gloo-coordinator.patch /ctx/overlay/docker/scripts/patches/dcp-save-gloo-coordinator.patch - git clone https://github.com/RLinf/RLinf.git /ctx/overlay/RLinf - git -C /ctx/overlay/RLinf checkout b3bbabb1f461 - git clone https://github.com/RLinf/dreamzero.git /ctx/overlay/DreamZero - git -C /ctx/overlay/DreamZero checkout ab790c198fbc - volumeMounts: - - { name: ctx, mountPath: /ctx } - - { name: cm, mountPath: /cm } - containers: - - name: kaniko - image: ghcr.io/chainguard-forks/kaniko/executor:latest - args: - - --dockerfile=/ctx/overlay/Dockerfile - - --context=dir:///ctx/overlay - - --build-arg=BUILD_TARGET=embodied-libero - - --build-arg=RLINF_UPSTREAM_IMAGE=${ECR_URI}:upstream-embodied-libero - - --destination=${ECR_URI}:latest - - --cache=true - - --cache-repo=${ECR_URI}/cache - volumeMounts: - - { name: ctx, mountPath: /ctx } - volumes: - - name: ctx - persistentVolumeClaim: - claimName: fsx-claim - - name: cm - configMap: - name: dreamzero-build-context - optional: false From 289576c3bd4644b60a2e15019c9e7ef883fd9732 Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Fri, 19 Jun 2026 16:13:40 -0500 Subject: [PATCH 15/45] chore(dreamzero): drop stray RLinf-on-EKS references from ported test case Remove source-repo references that leaked into the upstream port: - install_extras.sh / eval config: comment wording - dreamzero-wam{,-inference}.drawio + .svg: drop editor agent metadata --- .../pytorch/dreamzero/diagrams/dreamzero-wam-inference.drawio | 2 +- 3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam.drawio | 2 +- .../pytorch/dreamzero/diagrams/dreamzero-wam.drawio.svg | 4 ++-- .../pytorch/dreamzero/docker/scripts/install_extras.sh | 2 +- .../libero/scripts/libero_spatial_eval_dreamzero_14b.yaml | 2 +- 5 files changed, 6 insertions(+), 6 deletions(-) diff --git a/3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam-inference.drawio b/3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam-inference.drawio index 9dd13aaa7..86db1c72b 100644 --- a/3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam-inference.drawio +++ b/3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam-inference.drawio @@ -1,4 +1,4 @@ - + diff --git a/3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam.drawio b/3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam.drawio index 3b4402961..39f78ecb9 100644 --- a/3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam.drawio +++ b/3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam.drawio @@ -1,4 +1,4 @@ - + diff --git a/3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam.drawio.svg b/3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam.drawio.svg index f944e7101..3df0e2085 100644 --- a/3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam.drawio.svg +++ b/3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam.drawio.svg @@ -1,3 +1,3 @@ version https://git-lfs.github.com/spec/v1 -oid sha256:2a2190a15bbe23ea91b993b73f5d0b2a8a434b42846ba323dafb60fbabc5643d -size 716235 +oid sha256:8086115193e7fc255842f94d548de37952985f96266d56f6f7c08e59bac595f2 +size 716204 diff --git a/3.test_cases/pytorch/dreamzero/docker/scripts/install_extras.sh b/3.test_cases/pytorch/dreamzero/docker/scripts/install_extras.sh index 9a70bc5e2..0708f7061 100755 --- a/3.test_cases/pytorch/dreamzero/docker/scripts/install_extras.sh +++ b/3.test_cases/pytorch/dreamzero/docker/scripts/install_extras.sh @@ -2,7 +2,7 @@ # Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. # SPDX-License-Identifier: MIT-0 # ============================================================================= -# Install optional extras for RLinf-on-EKS container image +# Install optional extras for the DreamZero container image # # Called by Dockerfile with EXTRAS arg (comma-separated list). # Each extra is a self-contained install block. diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/libero_spatial_eval_dreamzero_14b.yaml b/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/libero_spatial_eval_dreamzero_14b.yaml index 84fadf62a..c36349b53 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/libero_spatial_eval_dreamzero_14b.yaml +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/libero_spatial_eval_dreamzero_14b.yaml @@ -2,7 +2,7 @@ # SPDX-License-Identifier: MIT-0 # libero_spatial_eval_dreamzero_14b.yaml # ============================================================================= -# DreamZero 14B LIBERO-Spatial simulator eval config (RLinf-on-EKS variant). +# DreamZero 14B LIBERO-Spatial simulator eval config (EKS variant). # # WHY THIS FILE EXISTS: # Upstream RLinf only ships a *5B* embodied eval config From 3f77130fe32582083d4da32beea577e5478bbf6d Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Fri, 19 Jun 2026 22:11:02 -0500 Subject: [PATCH 16/45] fix(dreamzero): make Dockerfile buildable + simplify build context First successful build of the test-case image (validated L1 CodeBuild + L2 container test, 10/10, on p5en.48xlarge): - Move RLINF_UPSTREAM_IMAGE ARG to global scope (before the first FROM). A per-stage ARG is invisible to a later FROM and resolves blank under BuildKit ("base name should not be blank"), which made the Dockerfile unbuildable by CodeBuild and docker buildx alike. - Drop the EXTRAS framework + install_extras.sh: its only value was an editable RLinf install into venvs the DreamZero workflow never uses; RLinf is imported from cwd (/workspace/RLinf), so the install was dead weight. Removes the misleading single-plugin 'extras' abstraction. - Drop run_training_eks.sh: the generic launcher is unused by this DreamZero-only test case (no manifest references it). - Relocate dcp-save-gloo-coordinator.patch to the test-case root and remove the empty docker/scripts/patches/ tree; COPY *.patch instead. - Correct DCP-fix comments: the sync dcp.save path is byte-identical through >= torch 2.8, so the patch is permanently required (the prior 'torch 2.7+ fixes this' claim was wrong). --- 3.test_cases/pytorch/dreamzero/Dockerfile | 73 +++---- 3.test_cases/pytorch/dreamzero/README.md | 2 +- .../dreamzero/dcp-save-gloo-coordinator.patch | 44 +++++ .../docker/scripts/install_extras.sh | 86 --------- .../patches/dcp-save-gloo-coordinator.patch | 38 ---- .../docker/scripts/run_training_eks.sh | 178 ------------------ .../dreamzero/kubernetes/libero/README.md | 2 +- 7 files changed, 77 insertions(+), 346 deletions(-) create mode 100644 3.test_cases/pytorch/dreamzero/dcp-save-gloo-coordinator.patch delete mode 100755 3.test_cases/pytorch/dreamzero/docker/scripts/install_extras.sh delete mode 100644 3.test_cases/pytorch/dreamzero/docker/scripts/patches/dcp-save-gloo-coordinator.patch delete mode 100644 3.test_cases/pytorch/dreamzero/docker/scripts/run_training_eks.sh diff --git a/3.test_cases/pytorch/dreamzero/Dockerfile b/3.test_cases/pytorch/dreamzero/Dockerfile index 3098d7fdd..692ea35b0 100644 --- a/3.test_cases/pytorch/dreamzero/Dockerfile +++ b/3.test_cases/pytorch/dreamzero/Dockerfile @@ -23,12 +23,20 @@ # UPSTREAM_REF) and DreamZero (pinned DREAMZERO_REF), builds the upstream # embodied-libero image as stage 1, then builds this EFA overlay as stage 2 and # pushes to your ECR. The build context is this test-case dir and must contain -# ./RLinf, ./DreamZero, and ./docker (scripts + patches). +# ./RLinf, ./DreamZero, and the *.patch file(s) applied to the RLinf source. # ============================================================================= # ---- Stage 1: Build upstream RLinf image ---- ARG BUILD_TARGET=embodied-maniskill_libero ARG UPSTREAM_DOCKERFILE=docker/Dockerfile +# RLINF_UPSTREAM_IMAGE points stage 2 at the stage-1 image. Declared in GLOBAL +# scope (before the first FROM) because an ARG used in a later FROM must be +# global -- a per-stage ARG is not visible to FROM and would resolve blank +# (BuildKit: "base name should not be blank"). The docker buildx path +# (setup/build-push.sh) and CodeBuild both build stage 1 as the local tag +# rlinf-upstream-${BUILD_TARGET}; override this arg to pull stage 1 from a +# registry instead. +ARG RLINF_UPSTREAM_IMAGE=rlinf-upstream-${BUILD_TARGET} FROM rlinf-upstream-${BUILD_TARGET} AS upstream # This FROM is a placeholder -- the actual upstream build happens in CodeBuild @@ -43,11 +51,6 @@ FROM rlinf-upstream-${BUILD_TARGET} AS upstream # ---- Stage 2: EFA networking overlay ---- # Layer EFA, GDRCopy, NCCL, and OpenMPI onto the upstream RLinf image. # Recipe proven in nccl-tests/Dockerfile on nvidia/cuda base images. - -# RLINF_UPSTREAM_IMAGE points stage 2 at the stage-1 image. The docker buildx -# path (setup/build-push.sh) uses the default local tag built immediately before; -# override it to pull a stage-1 image from a registry instead. -ARG RLINF_UPSTREAM_IMAGE=rlinf-upstream-${BUILD_TARGET} FROM ${RLINF_UPSTREAM_IMAGE} ARG GDRCOPY_VERSION=v2.5.1 @@ -178,57 +181,43 @@ ENV TOKENIZERS_PARALLELISM=true ENV TORCH_NCCL_AVOID_RECORD_STREAMS=1 ENV NCCL_DEBUG=WARN -# -- Copy EKS-specific scripts -- -COPY docker/scripts/ /workspace/eks/scripts/ +# -- Copy EKS hotfix patches (applied to the RLinf source below) -- +COPY *.patch /workspace/eks/patches/ # ============================================================================= -# Optional extras (controlled by EXTRAS build arg) -# -# EXTRAS is a comma-separated list of packages to install on top of the EFA -# overlay. This allows the same Dockerfile to produce images for different -# examples without rebuilding the expensive EFA/NCCL layers. +# Copy source repos into the build context. # -# Supported values: -# rlinf - RLinf source (required for all training examples) -# -# DreamZero note: the DreamZero `groot` package is provided by the -# COPY DreamZero/ tree on PYTHONPATH (DREAMZERO_PATH). Its `dreamzero` venv is -# built by the upstream embodied-libero target (RLinf PR #1272), so this overlay -# no longer rebuilds it. -# -# Examples: -# --build-arg EXTRAS=rlinf (ManiSkill, LIBERO, pi0, DreamZero) +# RLinf is always available (cloned in pre_build). RLinf is NOT pip-installed: +# the launchers run from /workspace/RLinf (cwd on sys.path), and the upstream +# image deliberately uses --no-install-project. DreamZero (groot package) is +# always cloned by the buildspec (pre_build) and made available on PYTHONPATH +# via DREAMZERO_PATH; its `dreamzero` venv is built by the upstream +# embodied-libero target (RLinf PR #1272), so this overlay does not rebuild it. # ============================================================================= -ARG EXTRAS="rlinf" - -# Copy source repos into build context. -# RLinf is always available (cloned in pre_build). -# DreamZero (groot package) is always cloned by the buildspec (pre_build) and -# made available on PYTHONPATH via DREAMZERO_PATH; it is no longer gated by EXTRAS. COPY RLinf/ /workspace/RLinf/ COPY DreamZero/ /workspace/DreamZero/ # Apply EKS hotfix patches to the upstream RLinf source. # dcp-save-gloo-coordinator.patch: pass a CPU/gloo process group to dcp.save so # its post-write finalization object-broadcast runs over gloo instead of CUDA/NCCL. -# On torch 2.6 the default-PG (NCCL) broadcast of the multi-MB finalization object -# races with NCCL comm teardown at the end of a long (100s+ GB) checkpoint write, -# leaving non-coordinator ranks with an all-zero buffer -> UnpicklingError, AFTER -# every shard + .metadata are already on disk (reproduced + validated on 2x p5en; -# torch 2.7+ fixes this upstream the same way). Must be baked into the image because -# Ray worker actors on every node import from /workspace/RLinf. git is present -# (installed in the EFA stage). Patches are validated to apply against the pinned -# UPSTREAM_REF; the build fails loudly if a patch no longer applies. +# dcp.save's broadcast_object_list picks its transport from the PG backend (CUDA +# for NCCL, CPU for gloo); with the default PG it runs on CUDA and races with NCCL +# comm teardown at the end of a long (100s+ GB) checkpoint write, leaving +# non-coordinator ranks with an all-zero buffer -> UnpicklingError, AFTER every +# shard + .metadata are already on disk (reproduced + validated on 2x p5en). +# NOTE: upgrading torch does NOT remove the need for this patch -- the synchronous +# dcp.save / _save_state_dict path is byte-identical through at least torch 2.8 and +# always broadcasts over the default PG (only async_save requires a CPU/gloo +# backend). Must be baked into the image because Ray worker actors on every node +# import from /workspace/RLinf. git is present (installed in the EFA stage). +# Patches are validated to apply against the pinned UPSTREAM_REF; the build fails +# loudly if a patch no longer applies. RUN cd /workspace/RLinf && \ - for p in /workspace/eks/scripts/patches/*.patch; do [ -e "$p" ] || continue; \ + for p in /workspace/eks/patches/*.patch; do [ -e "$p" ] || continue; \ echo "Applying patch: ${p}"; \ git apply --verbose "${p}" || patch -p1 < "${p}"; \ done -# Install extras based on EXTRAS arg -RUN chmod +x /workspace/eks/scripts/install_extras.sh && \ - /workspace/eks/scripts/install_extras.sh "${EXTRAS}" - # ============================================================================= # Security hardening / image hygiene (runs last so it cleans everything above) # diff --git a/3.test_cases/pytorch/dreamzero/README.md b/3.test_cases/pytorch/dreamzero/README.md index b8994f80a..60a1dbcd8 100644 --- a/3.test_cases/pytorch/dreamzero/README.md +++ b/3.test_cases/pytorch/dreamzero/README.md @@ -24,7 +24,7 @@ model code). ``` dreamzero/ ├── Dockerfile # two-stage RLinf + EFA overlay image -├── docker/ # build-context helpers (install_extras, DCP-save patch) +├── dcp-save-gloo-coordinator.patch # DCP checkpoint fix, applied to RLinf at build time ├── diagrams/ # WAM + infra topology (draw.io + SVG) └── kubernetes/libero/ # the EKS recipe (RayJob SFT + eval) — see its README ``` diff --git a/3.test_cases/pytorch/dreamzero/dcp-save-gloo-coordinator.patch b/3.test_cases/pytorch/dreamzero/dcp-save-gloo-coordinator.patch new file mode 100644 index 000000000..20acdc46b --- /dev/null +++ b/3.test_cases/pytorch/dreamzero/dcp-save-gloo-coordinator.patch @@ -0,0 +1,44 @@ +--- a/rlinf/hybrid_engines/fsdp/strategy/base.py ++++ b/rlinf/hybrid_engines/fsdp/strategy/base.py +@@ -233,9 +233,41 @@ + from torch.distributed import checkpoint as dcp + + dcp_save_path = os.path.join(save_path, "dcp_checkpoint") ++ # EKS root-cause fix (synchronous DCP save): dcp.save's ++ # post-write finalization runs all_reduce("write", write_data, ++ # finish_checkpoint), whose broadcast_object step pickles a ++ # multi-MB result object and broadcasts it via ++ # broadcast_object_list. That helper picks its transport device ++ # from the process group's backend (CUDA when NCCL, CPU when ++ # gloo); with no PG passed, dcp.save uses the default NCCL/CUDA ++ # group, so the object broadcast runs on CUDA and races with ++ # NCCL comm teardown at the end of a long (100s+ GB) checkpoint ++ # write -> non-coordinator ranks read an all-zero buffer -> ++ # `_pickle.UnpicklingError: invalid load key '\x00'`, AFTER ++ # every shard + .metadata are already on disk. ++ # NOTE: this is NOT fixed by upgrading torch -- the synchronous ++ # save() / _save_state_dict() path is byte-identical through at ++ # least torch 2.8 and always broadcasts over the default PG. ++ # (Only async_save requires/uses a CPU+gloo backend.) The fix is ++ # to pass a dedicated gloo process group to dcp.save so the ++ # finalization object-broadcast goes over CPU/gloo, immune to ++ # the CUDA/NCCL teardown race. ++ import datetime as _dt ++ ++ gloo_pg = getattr(cls, "_dcp_gloo_pg", None) ++ if gloo_pg is None and torch.distributed.is_initialized(): ++ try: ++ gloo_pg = torch.distributed.new_group( ++ backend="gloo", ++ timeout=_dt.timedelta(minutes=30), ++ ) ++ cls._dcp_gloo_pg = gloo_pg ++ except BaseException: ++ gloo_pg = None + dcp.save( + {"fsdp_checkpoint": training_state}, + checkpoint_id=dcp_save_path, ++ process_group=gloo_pg, + ) + + except BaseException as e: diff --git a/3.test_cases/pytorch/dreamzero/docker/scripts/install_extras.sh b/3.test_cases/pytorch/dreamzero/docker/scripts/install_extras.sh deleted file mode 100755 index 0708f7061..000000000 --- a/3.test_cases/pytorch/dreamzero/docker/scripts/install_extras.sh +++ /dev/null @@ -1,86 +0,0 @@ -#!/bin/bash -# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. -# SPDX-License-Identifier: MIT-0 -# ============================================================================= -# Install optional extras for the DreamZero container image -# -# Called by Dockerfile with EXTRAS arg (comma-separated list). -# Each extra is a self-contained install block. -# -# Supported extras: -# rlinf - RLinf source (editable install in all venvs) -# -# DreamZero deps are no longer installed here -- they come from upstream -# install.sh --env wan (stage 1) plus the DreamZero tree on PYTHONPATH. -# -# Usage (from Dockerfile): -# RUN /tmp/install_extras.sh "rlinf" -# ============================================================================= -set -euo pipefail - -EXTRAS="${1:-rlinf}" -VENVS="/opt/venv/openvla /opt/venv/openvla-oft /opt/venv/openpi" - -echo "=== Installing extras: ${EXTRAS} ===" - -install_in_venvs() { - local pkg_dir=$1 - local pkg_name=$2 - shift 2 - # Remaining args are extra pip packages to install - local extra_pkgs=("$@") - - for venv in ${VENVS}; do - if [ -d "$venv" ]; then - echo " Installing ${pkg_name} into $(basename $venv) venv..." - # Venv activate scripts reference PYTHONPATH/CPATH which may be unset - set +u - # shellcheck disable=SC1091 - . "$venv/bin/activate" - set -u - pip install --no-deps -e "$pkg_dir" 2>/dev/null - for pkg in "${extra_pkgs[@]}"; do - if [ -n "$pkg" ]; then - echo " Extra: $pkg" - MAX_JOBS=4 pip install --no-build-isolation "$pkg" 2>/dev/null || \ - echo " WARNING: $pkg install failed in $(basename $venv), skipping" - fi - done - set +u - deactivate - set -u - fi - done -} - -# Parse comma-separated EXTRAS into array -IFS=',' read -ra EXTRA_LIST <<< "$EXTRAS" - -for extra in "${EXTRA_LIST[@]}"; do - extra=$(echo "$extra" | tr -d ' ') # trim whitespace - case "$extra" in - rlinf) - echo "" - echo "--- Installing: RLinf source ---" - if [ -f /workspace/RLinf/pyproject.toml ]; then - echo " RLinf source found, installing..." - install_in_venvs /workspace/RLinf "RLinf" - else - echo " ERROR: RLinf source not found at /workspace/RLinf." - echo " Ensure buildspec clones RLinf and Dockerfile COPY's it." - exit 1 - fi - ;; - - "") - # Empty string from trailing comma, ignore - ;; - - *) - echo " WARNING: Unknown extra '$extra', skipping." - ;; - esac -done - -echo "" -echo "=== Extras installation complete ===" diff --git a/3.test_cases/pytorch/dreamzero/docker/scripts/patches/dcp-save-gloo-coordinator.patch b/3.test_cases/pytorch/dreamzero/docker/scripts/patches/dcp-save-gloo-coordinator.patch deleted file mode 100644 index fdbecf528..000000000 --- a/3.test_cases/pytorch/dreamzero/docker/scripts/patches/dcp-save-gloo-coordinator.patch +++ /dev/null @@ -1,38 +0,0 @@ ---- a/rlinf/hybrid_engines/fsdp/strategy/base.py -+++ b/rlinf/hybrid_engines/fsdp/strategy/base.py -@@ -233,9 +233,35 @@ - from torch.distributed import checkpoint as dcp - - dcp_save_path = os.path.join(save_path, "dcp_checkpoint") -+ # EKS root-cause fix (torch 2.6 DCP): dcp.save's post-write -+ # finalization runs all_reduce("write", write_data, -+ # finish_checkpoint), whose broadcast_object step pickles a -+ # multi-MB result object and broadcasts it over the process -+ # group. With the default NCCL/CUDA group, that object broadcast -+ # runs on CUDA and races with NCCL comm teardown at the end of a -+ # long (100s+ GB) checkpoint write -> non-coordinator ranks read -+ # an all-zero buffer -> `_pickle.UnpicklingError: invalid load -+ # key '\x00'`, AFTER every shard + .metadata are already on disk. -+ # torch 2.7+ fixes this by running DCP coordination on a CPU/gloo -+ # PG; backport that by passing a dedicated gloo process group to -+ # dcp.save so the object broadcast goes over CPU/gloo (immune to -+ # the CUDA/NCCL teardown race). -+ import datetime as _dt -+ -+ gloo_pg = getattr(cls, "_dcp_gloo_pg", None) -+ if gloo_pg is None and torch.distributed.is_initialized(): -+ try: -+ gloo_pg = torch.distributed.new_group( -+ backend="gloo", -+ timeout=_dt.timedelta(minutes=30), -+ ) -+ cls._dcp_gloo_pg = gloo_pg -+ except BaseException: -+ gloo_pg = None - dcp.save( - {"fsdp_checkpoint": training_state}, - checkpoint_id=dcp_save_path, -+ process_group=gloo_pg, - ) - - except BaseException as e: diff --git a/3.test_cases/pytorch/dreamzero/docker/scripts/run_training_eks.sh b/3.test_cases/pytorch/dreamzero/docker/scripts/run_training_eks.sh deleted file mode 100644 index 325ef4423..000000000 --- a/3.test_cases/pytorch/dreamzero/docker/scripts/run_training_eks.sh +++ /dev/null @@ -1,178 +0,0 @@ -#!/bin/bash -# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. -# SPDX-License-Identifier: MIT-0 -# ============================================================================= -# RLinf Training Launch Script for Amazon EKS -# -# Generic launcher that works with ANY RLinf embodiment config. -# The upstream multi-venv pattern is preserved: set VENV_NAME to select the -# model's Python environment, then CONFIG_NAME to select the Hydra config. -# -# Expects: -# - Container built from the two-stage Dockerfile (upstream RLinf + EFA) -# - Pre-trained model weights at MODEL_PATH (on shared storage) -# - ManiSkill/simulator assets available (via model-download job or baked in) -# - Shared storage at /fsx (FSx for Lustre PVC) -# - EFA environment variables set via pod spec -# -# Key environment variables (set via K8s manifest env): -# CONFIG_NAME - Hydra config name (e.g., maniskill_ppo_openvla_quickstart) -# VENV_NAME - Python venv to activate (e.g., openvla, openvla-oft, openpi) -# MODEL_PATH - Path to pre-trained model weights on shared storage -# CKPT_PATH - Path to write checkpoints -# NUM_GPUS - GPUs per node -# NUM_NODES - Number of nodes -# ============================================================================= -set -euo pipefail -set -x - -# --- Configuration (override via environment variables) --- -EXPERIMENT_NAME="${EXPERIMENT_NAME:-rlinf-eks}" -CONFIG_NAME="${CONFIG_NAME:-maniskill_ppo_openvla_quickstart}" -VENV_NAME="${VENV_NAME:-openvla}" - -MODEL_PATH="${MODEL_PATH:-/fsx/models/openvla-7b-rlvla-warmup}" -CKPT_PATH="${CKPT_PATH:-/fsx/checkpoints}" - -NUM_GPUS="${NUM_GPUS:-8}" -NUM_NODES="${NUM_NODES:-1}" - -# Component placement string (e.g., "0-7" for 8 GPUs) -GPU_RANGE="0-$((NUM_GPUS - 1))" -COMPONENT_PLACEMENT="${COMPONENT_PLACEMENT:-actor,env,rollout: ${GPU_RANGE}}" - -# --- Activate the correct venv --- -# Upstream RLinf images use multi-venv pattern with switch_env utility. -# Each model (openvla, openvla-oft, openpi, gr00t, etc.) has its own venv. -UV_PATH="${UV_PATH:-/opt/venv}" - -# Pre-set variables that venv activate scripts may reference but are not -# guaranteed to exist in container environments (avoids "unbound variable" -# errors under set -u). -export PYTHONPATH="${PYTHONPATH:-}" - -if [ -f "${UV_PATH}/${VENV_NAME}/bin/activate" ]; then - echo "Activating venv: ${VENV_NAME}" - source "${UV_PATH}/${VENV_NAME}/bin/activate" -elif [ -f "/usr/local/bin/switch_env" ]; then - echo "Activating venv via switch_env: ${VENV_NAME}" - source switch_env "${VENV_NAME}" -else - echo "WARNING: No venv found for ${VENV_NAME}, using system Python" -fi - -echo "Python: $(which python3)" -echo "PyTorch: $(python3 -c 'import torch; print(torch.__version__)' 2>/dev/null || echo 'not found')" - -# --- Environment (defaults set in Dockerfile, override via pod spec) --- -export NCCL_DEBUG="${NCCL_DEBUG:-WARN}" -export PYTORCH_CUDA_ALLOC_CONF="${PYTORCH_CUDA_ALLOC_CONF:-expandable_segments:True}" -export TOKENIZERS_PARALLELISM="${TOKENIZERS_PARALLELISM:-true}" - -# Ensure libcuda.so is discoverable by subprocesses (Ray env offload workers). -# The NVIDIA runtime mounts it at /usr/lib64 but that may not be in LD_LIBRARY_PATH. -export LD_LIBRARY_PATH="/usr/lib64:${LD_LIBRARY_PATH:-}" - -# ManiSkill / MuJoCo headless rendering -export MUJOCO_GL="${MUJOCO_GL:-osmesa}" -export PYOPENGL_PLATFORM="${PYOPENGL_PLATFORM:-osmesa}" -export NVIDIA_DRIVER_CAPABILITIES="${NVIDIA_DRIVER_CAPABILITIES:-all}" - -# EFA (set in pod spec, but provide defaults) -export FI_PROVIDER="${FI_PROVIDER:-efa}" -export FI_EFA_USE_DEVICE_RDMA="${FI_EFA_USE_DEVICE_RDMA:-1}" -export FI_EFA_FORK_SAFE="${FI_EFA_FORK_SAFE:-1}" - -# --- Link simulator assets if available --- -# The upstream image provides link_assets for ManiSkill/SAPIEN symlinks -if [ -x "/usr/local/bin/link_assets" ]; then - link_assets -fi - -# --- Pre-step: Verify model weights --- -if [ -n "${MODEL_PATH}" ] && [ "${MODEL_PATH}" != "none" ]; then - if [ ! -f "$MODEL_PATH/config.json" ] && [ ! -f "$MODEL_PATH/model.safetensors" ]; then - echo "ERROR: Model weights not found at $MODEL_PATH" - echo "Expected config.json or model.safetensors. Run the model-download job first." - exit 1 - fi -fi - -# Clear stale Python bytecode from previous runs (FSx is persistent storage) -find "$CKPT_PATH" -name "__pycache__" -exec rm -rf {} + 2>/dev/null || true - -# Clear HuggingFace transformers dynamic module cache -rm -rf /root/.cache/huggingface/modules/transformers_modules/ 2>/dev/null || true - -# --- Launch training --- -echo "=== Launching RLinf training ===" -echo "Config: ${CONFIG_NAME}" -echo "Venv: ${VENV_NAME}" -echo "Model: ${MODEL_PATH}" -echo "GPUs: ${NUM_GPUS} x ${NUM_NODES} nodes" - -cd /workspace/RLinf - -# Set EMBODIED_PATH for Hydra config interpolation (used in upstream configs -# to resolve relative paths to examples/embodiment/). -export EMBODIED_PATH="${EMBODIED_PATH:-/workspace/RLinf/examples/embodiment}" - -# Build Hydra override args. -# Start with model paths and cluster config, then append any user-supplied overrides. -# HYDRA_OVERRIDES env var allows callers (e.g., validation harness) to inject -# additional overrides like "runner.max_steps=1" without modifying this script. -HYDRA_ARGS="" -if [ -n "${MODEL_PATH}" ] && [ "${MODEL_PATH}" != "none" ]; then - HYDRA_ARGS="actor.model.model_path=${MODEL_PATH} rollout.model.model_path=${MODEL_PATH}" -fi -HYDRA_ARGS="${HYDRA_ARGS} cluster.num_nodes=${NUM_NODES}" - -# Append user-supplied overrides (e.g., HYDRA_OVERRIDES="runner.max_steps=1") -if [ -n "${HYDRA_OVERRIDES:-}" ]; then - echo "Extra Hydra overrides: ${HYDRA_OVERRIDES}" - HYDRA_ARGS="${HYDRA_ARGS} ${HYDRA_OVERRIDES}" -fi - -# Support YAML config override file for complex keys (e.g., component_placement -# with commas in the key name that Hydra CLI cannot parse). -# Set HYDRA_CONFIG_FILE to a YAML file path; its contents will be patched into -# the base config file before launching training. The container is ephemeral, -# so in-place modification is safe. -if [ -n "${HYDRA_CONFIG_FILE:-}" ] && [ -f "${HYDRA_CONFIG_FILE}" ]; then - CONFIG_FILE="examples/embodiment/config/${CONFIG_NAME}.yaml" - echo "Patching config with override file: ${HYDRA_CONFIG_FILE}" - python3 -c " -import yaml, sys -with open('${CONFIG_FILE}') as f: - base = yaml.safe_load(f) -with open('${HYDRA_CONFIG_FILE}') as f: - override = yaml.safe_load(f) - -def deep_merge(base, override): - for k, v in override.items(): - if k in base and isinstance(base[k], dict) and isinstance(v, dict): - deep_merge(base[k], v) - else: - base[k] = v - -deep_merge(base, override) -with open('${CONFIG_FILE}', 'w') as f: - yaml.dump(base, f, default_flow_style=False, sort_keys=False) -print(f'Patched {len(override)} top-level keys into ${CONFIG_FILE}') -" -fi - -LOG_DIR="${CKPT_PATH}/${EXPERIMENT_NAME}" -HYDRA_ARGS="${HYDRA_ARGS} runner.logger.log_path=${LOG_DIR}" - -echo "Hydra args: ${HYDRA_ARGS}" - -# Call train_embodied_agent.py directly (not run_embodiment.sh) so we can pass -# arbitrary Hydra overrides. run_embodiment.sh does not forward extra args. -# --config-path is relative to the script's directory (examples/embodiment/), -# so use just "config" not the full path from repo root. -# shellcheck disable=SC2086 -python3 examples/embodiment/train_embodied_agent.py \ - --config-path config \ - --config-name "${CONFIG_NAME}" \ - ${HYDRA_ARGS} diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md index d2b704ad0..c4d2ec53d 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md @@ -469,7 +469,7 @@ config: |---------|-------|-----| | Pods crash with `ray: command not found`, or inline `${...}` vars render empty | Manifest rendered with unrestricted `envsubst`, clobbering inline shell vars | Always render manifests that embed shell with the **restricted** allow-list: `envsubst '${ECR_URI} ${NAMESPACE}' < ...`. | | `ray: command not found` on Ray head/worker/submitter | KubeRay runs `ray` non-interactively (`~/.bashrc` not sourced); `ray` lives in the `dreamzero` venv | Already handled in `dreamzero-sft.yaml`: the venv `bin` is prepended to `PATH` on the head/worker `env` **and** on the `submitterPodTemplate`. Don't strip those `PATH` entries. | -| SFT crashes near checkpoint save: `UnpicklingError: invalid load key '\x00'` in `broadcast_object_list`, after shards are written | torch 2.6 DCP broadcasts its finalization object over the default NCCL/CUDA PG, which races with NCCL teardown at the end of a long write | Fixed by the in-image `dcp-save-gloo-coordinator.patch` (routes the broadcast over a CPU/gloo PG, like torch 2.7+). If you somehow hit this, the **on-disk checkpoint is still valid and convertible** — proceed to step 5 (convert). | +| SFT crashes near checkpoint save: `UnpicklingError: invalid load key '\x00'` in `broadcast_object_list`, after shards are written | `dcp.save`'s finalization broadcast runs over the default NCCL/CUDA PG and races with NCCL teardown at the end of a long write (not torch-version-specific — the sync `dcp.save` path is unchanged through ≥ torch 2.8) | Fixed by the in-image `dcp-save-gloo-coordinator.patch`, which passes a dedicated gloo PG so the broadcast runs over CPU/gloo. If you somehow hit this, the **on-disk checkpoint is still valid and convertible** — proceed to step 5 (convert). | | Convert fails with `EOFError` / `inline_container.cc unexpected pos`, corrupt shards | FSx filled up during the SFT run; `torch.save` was truncated mid-write | Ensure **≥250 GB free** on FSx before SFT (a 14B DCP checkpoint is ~140–206 GB). Free space, re-run SFT. | | SFT fails with `KeyError: embodiment_tag 'libero_sim' not found` | The DROID checkpoint only bundles `oxe_droid` metadata | Run step 3 (`generate-metadata.yaml`) to produce `/fsx/models/metadata-libero.json` before SFT/eval. | | Metadata generation or transforms behave wrongly / schema errors | Wrong dataset (`lerobot/libero` has a different `observation.state`/`action` schema) | The dataset **must** be `physical-intelligence/libero`. Re-stage step 2. | From 58477a8c9dc3434df305557c7e51f5c9e72b9ab0 Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Fri, 19 Jun 2026 22:44:59 -0500 Subject: [PATCH 17/45] chore(dreamzero): move build-push.sh out of one-file setup/ dir Relocate kubernetes/libero/setup/build-push.sh -> kubernetes/libero/build-push.sh and drop the single-file setup/ directory. Matches the sibling openvla-oft/kubernetes/libero/ layout (helper scripts flat in libero/) and colocates the script with the env_vars it sources. - ROOT path ../../.. -> ../.. (one level shallower); verified it still resolves to the test-case root (the buildx build context). - Usage comment: source ../env_vars -> source ./env_vars (now same dir). - Update references in README.md (root + libero walkthrough) and Dockerfile comments. --- 3.test_cases/pytorch/dreamzero/Dockerfile | 4 ++-- 3.test_cases/pytorch/dreamzero/README.md | 2 +- .../pytorch/dreamzero/kubernetes/libero/README.md | 13 +++++-------- .../kubernetes/libero/{setup => }/build-push.sh | 6 +++--- 4 files changed, 11 insertions(+), 14 deletions(-) rename 3.test_cases/pytorch/dreamzero/kubernetes/libero/{setup => }/build-push.sh (88%) diff --git a/3.test_cases/pytorch/dreamzero/Dockerfile b/3.test_cases/pytorch/dreamzero/Dockerfile index 692ea35b0..ad2ee260d 100644 --- a/3.test_cases/pytorch/dreamzero/Dockerfile +++ b/3.test_cases/pytorch/dreamzero/Dockerfile @@ -19,7 +19,7 @@ # - reason (Agentic RL, vLLM/SGLang) # - ... (15 targets total, see upstream docker/Dockerfile) # -# Build: use kubernetes/libero/setup/build-push.sh, which clones RLinf (pinned +# Build: use kubernetes/libero/build-push.sh, which clones RLinf (pinned # UPSTREAM_REF) and DreamZero (pinned DREAMZERO_REF), builds the upstream # embodied-libero image as stage 1, then builds this EFA overlay as stage 2 and # pushes to your ECR. The build context is this test-case dir and must contain @@ -33,7 +33,7 @@ ARG UPSTREAM_DOCKERFILE=docker/Dockerfile # scope (before the first FROM) because an ARG used in a later FROM must be # global -- a per-stage ARG is not visible to FROM and would resolve blank # (BuildKit: "base name should not be blank"). The docker buildx path -# (setup/build-push.sh) and CodeBuild both build stage 1 as the local tag +# (build-push.sh) and CodeBuild both build stage 1 as the local tag # rlinf-upstream-${BUILD_TARGET}; override this arg to pull stage 1 from a # registry instead. ARG RLINF_UPSTREAM_IMAGE=rlinf-upstream-${BUILD_TARGET} diff --git a/3.test_cases/pytorch/dreamzero/README.md b/3.test_cases/pytorch/dreamzero/README.md index 60a1dbcd8..2faca96ad 100644 --- a/3.test_cases/pytorch/dreamzero/README.md +++ b/3.test_cases/pytorch/dreamzero/README.md @@ -60,7 +60,7 @@ checkpoint yields `eval/success_once = 0.0`, which is **expected**; real accurac requires a multi-step SFT run (raise `runner.max_steps`). The image is built with the **local `docker buildx`** two-stage build and pushed -to ECR (see [`kubernetes/libero/setup/build-push.sh`](kubernetes/libero/setup/build-push.sh)). +to ECR (see [`kubernetes/libero/build-push.sh`](kubernetes/libero/build-push.sh)). ## References diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md index c4d2ec53d..fa604045d 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md @@ -273,15 +273,13 @@ libfabric 2.4.0 / aws-ofi-nccl 1.18.0 and the DCP-save gloo-coordinator patch. T external `groot` package is cloned from `github.com/RLinf/dreamzero.git` and placed on `PYTHONPATH` via `DREAMZERO_PATH=/workspace/DreamZero`. -Build the image with **local Docker + buildx**. Run from the `setup/` directory; -it reads `ECR_URI`, `AWS_REGION`, `UPSTREAM_REF`, and `DREAMZERO_REF` from -`env_vars`: +Build the image with **local Docker + buildx**. Run from the `kubernetes/libero/` +directory; it reads `ECR_URI`, `AWS_REGION`, `UPSTREAM_REF`, and `DREAMZERO_REF` +from `env_vars`: ```bash -cd setup -source ../env_vars +source ./env_vars ./build-push.sh -cd .. ``` This clones the pinned `RLinf` and `dreamzero` sources, builds stage 1 @@ -412,6 +410,7 @@ kubernetes/libero/ ├── README.md # This walkthrough ├── env_vars.example # Copy to env_vars and `source` it ├── secret.example.yaml # OPTIONAL hf-token Secret (gated repos only) +├── build-push.sh # Local buildx two-stage build + push to ECR ├── model-download.yaml # Job: stage DreamZero-DROID + umt5-xxl + libero dataset ├── generate-metadata.yaml # Job: libero_sim normalization metadata.json ├── dreamzero-sft.yaml # KubeRay RayJob: 2-node FSDP2 SFT @@ -422,8 +421,6 @@ kubernetes/libero/ │ ├── convert_checkpoint.sh # DCP -> .pt conversion launcher │ ├── run_dreamzero_eval_eks.sh # LIBERO simulator eval launcher │ └── libero_spatial_eval_dreamzero_14b.yaml # 14B eval config (upstream ships 5B only) -├── setup/ -│ └── build-push.sh # Local buildx two-stage build + push to ECR └── storage/ ├── pvc-fsx-lustre-dynamic.yaml # OPTIONAL: dynamically provision fsx-claim └── pv-fsx-lustre-static.yaml # OPTIONAL: bind fsx-claim to an existing FSx diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/setup/build-push.sh b/3.test_cases/pytorch/dreamzero/kubernetes/libero/build-push.sh similarity index 88% rename from 3.test_cases/pytorch/dreamzero/kubernetes/libero/setup/build-push.sh rename to 3.test_cases/pytorch/dreamzero/kubernetes/libero/build-push.sh index 0a1b6130a..7b17e3d59 100755 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/setup/build-push.sh +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/build-push.sh @@ -6,7 +6,7 @@ # EFA overlay) and push to your ECR. Requires docker buildx + AWS CLI logged in. # # Usage: -# source ../env_vars # sets ECR_URI, AWS_REGION, UPSTREAM_REF, DREAMZERO_REF +# source ./env_vars # sets ECR_URI, AWS_REGION, UPSTREAM_REF, DREAMZERO_REF # ./build-push.sh set -euo pipefail @@ -19,8 +19,8 @@ DREAMZERO_REF="${DREAMZERO_REF:-ab790c198fbc}" BUILD_TARGET="${BUILD_TARGET:-embodied-libero}" TAG="${TAG:-latest}" -# Test-case root (this script is at kubernetes/libero/setup/build-push.sh). -ROOT="$(cd "$(dirname "$0")/../../.." && pwd)" +# Test-case root (this script is at kubernetes/libero/build-push.sh). +ROOT="$(cd "$(dirname "$0")/../.." && pwd)" cd "$ROOT" echo "== ECR login ==" From a3b9e564aaac70321e9fbd31faca6a34401c3e6f Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Fri, 19 Jun 2026 23:50:16 -0500 Subject: [PATCH 18/45] docs(dreamzero): correct model/transfer claims to match sources Align the READMEs with the DreamZero paper, HF model card, and RLinf docs: - Parameter count in the intros: 16.48B -> 14B (the published headline; the README titles already say 14B). The measured ~16.48B instantiated-model figure is kept where it matters (FSDP sharding / OOM / VRAM sections). - Architecture phrasing: drop the unsourced 'shared causal self-attention space' for 'causal (autoregressive) ... via flow matching', grounded in the CausalWanModel class and the paper. - LIBERO framing: it is a manipulation benchmark on the same Franka arm as DROID, not a new embodiment, so drop 'new embodiment' / 'cross-embodiment transfer' and frame the sample around its real purpose (EKS deployment). - Add the upstream 5B LIBERO-Spatial accuracy (~96.7% success_once at step 18000) as evidence the recipe converges with sufficient steps. --- 3.test_cases/pytorch/dreamzero/README.md | 23 ++++++++++++------- .../dreamzero/kubernetes/libero/README.md | 10 ++++---- 2 files changed, 21 insertions(+), 12 deletions(-) diff --git a/3.test_cases/pytorch/dreamzero/README.md b/3.test_cases/pytorch/dreamzero/README.md index 2faca96ad..991931f50 100644 --- a/3.test_cases/pytorch/dreamzero/README.md +++ b/3.test_cases/pytorch/dreamzero/README.md @@ -3,14 +3,17 @@ # DreamZero LIBERO 14B SFT (World-Action Model) on Amazon EKS -DreamZero is a **16.48B-parameter World-Action Model (WAM)** — a Wan-based -video-diffusion Diffusion Transformer that *jointly* denoises future video frames -and robot actions in a shared causal self-attention space. This test case -continues supervised fine-tuning from the released +This test case is a worked example of **deploying DreamZero training on Amazon +EKS**. DreamZero is a **14B-parameter World-Action Model (WAM)** — a Wan-based, +causal (autoregressive) video diffusion transformer that *jointly* denoises +future video frames and robot actions via flow matching. Here it continues +supervised fine-tuning from the released [`GEAR-Dreams/DreamZero-DROID`](https://huggingface.co/GEAR-Dreams/DreamZero-DROID) -14B checkpoint onto a *new* embodiment (LIBERO) and evaluates it in the LIBERO -simulator on Amazon EKS — demonstrating cross-embodiment transfer and multi-node -**FSDP2** training over **EFA** with **KubeRay**. Upstream: +14B checkpoint on the **LIBERO** manipulation benchmark and evaluates it in the +LIBERO simulator — exercising multi-node **FSDP2** training over **EFA** with +**KubeRay**. The focus is the **EKS deployment mechanics** (image build, +multi-node EFA/NCCL, FSDP2 sharded checkpointing, sim eval), not a training-quality +or transfer result. Upstream: [github.com/RLinf/RLinf](https://github.com/RLinf/RLinf) (training framework) and [github.com/RLinf/dreamzero](https://github.com/RLinf/dreamzero) (the `groot` WAM model code). @@ -57,7 +60,11 @@ on CPU and consumed by the LIBERO simulator eval. This proves the *infrastructur and pipeline* (image build, multi-node EFA/NCCL, FSDP2 sharded checkpointing, DCP→`.pt` conversion, and in-sim eval) — **not** task accuracy. A 1-step checkpoint yields `eval/success_once = 0.0`, which is **expected**; real accuracy -requires a multi-step SFT run (raise `runner.max_steps`). +requires a multi-step SFT run (raise `runner.max_steps`). For reference, RLinf's +own LIBERO-Spatial SFT of the **5B** WAM +([`RLinf-DreamZero-WAN2.2-5B-LIBERO-SFT-Step18000`](https://huggingface.co/RLinf/RLinf-DreamZero-WAN2.2-5B-LIBERO-SFT-Step18000)) +reaches **~96.7% `success_once` by step 18000** ([RLinf docs](https://rlinf.readthedocs.io/en/latest/rst_source/examples/embodied/sft_dreamzero.html)), +confirming the recipe converges with sufficient steps. The image is built with the **local `docker buildx`** two-stage build and pushed to ECR (see [`kubernetes/libero/build-push.sh`](kubernetes/libero/build-push.sh)). diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md index fa604045d..58abcee13 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md @@ -3,9 +3,10 @@ # Continue SFT of DreamZero (14B World-Action Model) on LIBERO on Amazon EKS -DreamZero is a **16.48B-parameter World-Action Model (WAM)** — a Wan-based -video-diffusion Diffusion Transformer (DiT) that *jointly* denoises future video -frames and future robot actions in a shared causal self-attention space. The +DreamZero is a **14B-parameter World-Action Model (WAM)** — a Wan-based, +causal (autoregressive) video-diffusion Diffusion Transformer (DiT) that +*jointly* denoises future video frames and future robot actions via flow +matching. The model predicts both what will happen (video) and what to do (actions); the video prediction acts as a computational scaffold for action reasoning. @@ -13,7 +14,8 @@ This walkthrough packages the canonical customer workflow as a set of reusable Kubernetes manifests that run on Amazon EKS: take the released [`GEAR-Dreams/DreamZero-DROID`](https://huggingface.co/GEAR-Dreams/DreamZero-DROID) 14B checkpoint (pretrained on DROID, a Franka arm), continue **supervised -fine-tuning (SFT)** on a *new* embodiment's data (LIBERO, `libero_sim`), then +fine-tuning (SFT)** on the **LIBERO** benchmark (also a Franka arm, in +simulation; `embodiment_tag: libero_sim`), then **evaluate the result in the LIBERO simulator** and render in-sim rollout videos. There is no native LIBERO 14B checkpoint upstream — warm-starting from DROID is the point. From b4d710919646f2b4e07e8fb78e49fa5450210dde Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Sat, 20 Jun 2026 13:03:50 -0500 Subject: [PATCH 19/45] docs(dreamzero): note real->sim domain gap + swap-in-your-own-data caveat Add a closing caveat to the intro: LIBERO is a simulation of the same Franka Panda arm DROID captures in the real world, so warm-starting onto LIBERO bridges a real->sim visual domain gap, and in practice users would substitute their own dataset for task-specific or cross-embodiment fine-tuning. Avoids the term 'negative transfer' (the RLinf docs recommend warm-starting from the released checkpoint, i.e. the prior is beneficial). --- 3.test_cases/pytorch/dreamzero/README.md | 16 ++++++++++++---- 1 file changed, 12 insertions(+), 4 deletions(-) diff --git a/3.test_cases/pytorch/dreamzero/README.md b/3.test_cases/pytorch/dreamzero/README.md index 991931f50..ec3ec8239 100644 --- a/3.test_cases/pytorch/dreamzero/README.md +++ b/3.test_cases/pytorch/dreamzero/README.md @@ -3,7 +3,7 @@ # DreamZero LIBERO 14B SFT (World-Action Model) on Amazon EKS -This test case is a worked example of **deploying DreamZero training on Amazon +This test case is a working example of **deploying DreamZero training on Amazon EKS**. DreamZero is a **14B-parameter World-Action Model (WAM)** — a Wan-based, causal (autoregressive) video diffusion transformer that *jointly* denoises future video frames and robot actions via flow matching. Here it continues @@ -13,9 +13,17 @@ supervised fine-tuning from the released LIBERO simulator — exercising multi-node **FSDP2** training over **EFA** with **KubeRay**. The focus is the **EKS deployment mechanics** (image build, multi-node EFA/NCCL, FSDP2 sharded checkpointing, sim eval), not a training-quality -or transfer result. Upstream: -[github.com/RLinf/RLinf](https://github.com/RLinf/RLinf) (training framework) and -[github.com/RLinf/dreamzero](https://github.com/RLinf/dreamzero) (the `groot` WAM +or transfer result. Note that LIBERO is a **simulation** of the same Franka Emika +Panda arm that DROID captures in the **real world**, so warm-starting the DROID +checkpoint onto LIBERO must bridge a real→sim visual domain gap. LIBERO is used +here as a convenient public dataset + simulator to exercise the pipeline +end-to-end; in practice you would swap it for your own dataset (real or simulated) +for task-specific or cross-embodiment fine-tuning. + +Upstream: + +- [github.com/RLinf/RLinf](https://github.com/RLinf/RLinf) (training framework) +- [github.com/RLinf/dreamzero](https://github.com/RLinf/dreamzero) (the `groot` WAM model code). ## Architecture From 71c6f63aa455088b751edb5406942afa2955ce36 Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Sat, 20 Jun 2026 13:50:37 -0500 Subject: [PATCH 20/45] docs(dreamzero): link root README to detailed architecture; drop unused inference diagram - Root README Architecture section now links to the full topology + WAM component breakdown in kubernetes/libero/README.md#architecture (previously just a bare image with no pointer to the detail). - Remove dreamzero-wam-inference.drawio + .svg: it depicts the paper's closed-loop real-time inference path, which this SFT+eval test case does not cover. It was referenced by no README in either repo and was already flagged for removal in the RLinf-on-eks rearchitecture plan. --- 3.test_cases/pytorch/dreamzero/README.md | 4 + .../diagrams/dreamzero-wam-inference.drawio | 144 ------------------ .../dreamzero-wam-inference.drawio.svg | 3 - 3 files changed, 4 insertions(+), 147 deletions(-) delete mode 100644 3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam-inference.drawio delete mode 100644 3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam-inference.drawio.svg diff --git a/3.test_cases/pytorch/dreamzero/README.md b/3.test_cases/pytorch/dreamzero/README.md index ec3ec8239..964274eb9 100644 --- a/3.test_cases/pytorch/dreamzero/README.md +++ b/3.test_cases/pytorch/dreamzero/README.md @@ -30,6 +30,10 @@ model code). ![DreamZero WAM](diagrams/dreamzero-wam.drawio.svg) +For the full breakdown — the KubeRay/FSDP2 training topology and a component-level +walkthrough of the World-Action Model — see +[**Architecture** in the walkthrough](kubernetes/libero/README.md#architecture). + ## Layout ``` diff --git a/3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam-inference.drawio b/3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam-inference.drawio deleted file mode 100644 index 86db1c72b..000000000 --- a/3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam-inference.drawio +++ /dev/null @@ -1,144 +0,0 @@ - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - diff --git a/3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam-inference.drawio.svg b/3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam-inference.drawio.svg deleted file mode 100644 index 71c127d81..000000000 --- a/3.test_cases/pytorch/dreamzero/diagrams/dreamzero-wam-inference.drawio.svg +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:12cab47be2b07cfd3e13db1f54293b5e68ec5e4ed1fd250dfc4cca74f28d2f31 -size 732106 From d0c16652f8094bb89fc06e2a4b8a6b77059449b5 Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Sat, 20 Jun 2026 14:08:33 -0500 Subject: [PATCH 21/45] docs(dreamzero): fix broken 1.architectures link + reframe Prerequisites - Fix the relative path to 1.architectures/4.amazon-eks: it needs ../../../ (dreamzero -> pytorch -> 3.test_cases -> root), not ../../ which dead-ends in 3.test_cases/. Both the Prerequisites and References links were wrong; the sibling openvla-oft uses the correct depth. - Reframe Prerequisites around the nodes, not the provisioning mechanism. The RayJob is fixed-size (head 1 + worker replicas/min/max = 1), so GPU autoscaling is a convenience (on-demand p5en provisioning + scale-down), not a requirement -- a static managed node group or Capacity Block works identically. The only Karpenter-ism shipped is the karpenter.sh/do-not-disrupt annotation, which is ignored on non-Karpenter clusters. --- 3.test_cases/pytorch/dreamzero/README.md | 11 +++++++---- 1 file changed, 7 insertions(+), 4 deletions(-) diff --git a/3.test_cases/pytorch/dreamzero/README.md b/3.test_cases/pytorch/dreamzero/README.md index 964274eb9..671dfb040 100644 --- a/3.test_cases/pytorch/dreamzero/README.md +++ b/3.test_cases/pytorch/dreamzero/README.md @@ -52,9 +52,12 @@ node**, and **FSx for Lustre with ≥250 GB free** (a 14B FSDP DCP checkpoint is ## Prerequisites -An Amazon EKS cluster with GPU autoscaling (Karpenter), EFA networking, the -KubeRay operator, and FSx for Lustre shared storage. See -[`../../1.architectures/4.amazon-eks`](../../1.architectures/4.amazon-eks) for +An Amazon EKS cluster that can schedule **2× `p5en.48xlarge`** (8× H200 + EFA +each) — provisioned however you like (a static managed node group, a Capacity +Block reservation, or on-demand autoscaling such as Karpenter; the workload is +fixed-size, so autoscaling is a convenience, not a requirement) — plus EFA +networking, the KubeRay operator, and FSx for Lustre shared storage. See +[`1.architectures/4.amazon-eks`](../../../1.architectures/4.amazon-eks) for cluster setup. The detailed prerequisite checklist lives in the walkthrough below. ## Full walkthrough @@ -86,7 +89,7 @@ to ECR (see [`kubernetes/libero/build-push.sh`](kubernetes/libero/build-push.sh) - RLinf training framework — [github.com/RLinf/RLinf](https://github.com/RLinf/RLinf) - DreamZero (`groot`) model code — [github.com/RLinf/dreamzero](https://github.com/RLinf/dreamzero) - DreamZero-DROID checkpoint — [huggingface.co/GEAR-Dreams/DreamZero-DROID](https://huggingface.co/GEAR-Dreams/DreamZero-DROID) -- EKS cluster architectures — [`1.architectures/4.amazon-eks`](../../1.architectures/4.amazon-eks) +- EKS cluster architectures — [`1.architectures/4.amazon-eks`](../../../1.architectures/4.amazon-eks) ## Security From c0ed99bb3a3d51e1a91427c216f3e4dd3428186c Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Sat, 20 Jun 2026 14:52:36 -0500 Subject: [PATCH 22/45] docs(dreamzero): fix stale buildspec/CodeBuild references to build-push.sh This test case builds via kubernetes/libero/build-push.sh (docker buildx); it ships no buildspec.yml. Several comments still referenced CodeBuild / a buildspec (carryover from the RLinf-on-eks origin): - Dockerfile: the RLINF_UPSTREAM_IMAGE note and the stage-1 placeholder comment now describe the build-push.sh flow (and fix the local-build example's BUILD_TARGET: embodied-libero, not embodied-maniskill_libero); the 'cloned in pre_build / by the buildspec' notes now say build-push.sh. - dreamzero-eval.yaml: prerequisite note '(examples/buildspec.yml)' -> '(build-push.sh)'. Also improve the 1.architectures link text and target the section README. --- 3.test_cases/pytorch/dreamzero/Dockerfile | 27 ++++++++++--------- 3.test_cases/pytorch/dreamzero/README.md | 2 +- .../kubernetes/libero/dreamzero-eval.yaml | 2 +- 3 files changed, 16 insertions(+), 15 deletions(-) diff --git a/3.test_cases/pytorch/dreamzero/Dockerfile b/3.test_cases/pytorch/dreamzero/Dockerfile index ad2ee260d..1f49b824e 100644 --- a/3.test_cases/pytorch/dreamzero/Dockerfile +++ b/3.test_cases/pytorch/dreamzero/Dockerfile @@ -33,19 +33,19 @@ ARG UPSTREAM_DOCKERFILE=docker/Dockerfile # scope (before the first FROM) because an ARG used in a later FROM must be # global -- a per-stage ARG is not visible to FROM and would resolve blank # (BuildKit: "base name should not be blank"). The docker buildx path -# (build-push.sh) and CodeBuild both build stage 1 as the local tag +# (kubernetes/libero/build-push.sh) builds stage 1 as the local tag # rlinf-upstream-${BUILD_TARGET}; override this arg to pull stage 1 from a # registry instead. ARG RLINF_UPSTREAM_IMAGE=rlinf-upstream-${BUILD_TARGET} FROM rlinf-upstream-${BUILD_TARGET} AS upstream -# This FROM is a placeholder -- the actual upstream build happens in CodeBuild -# (see buildspec.yml). CodeBuild builds the upstream image first, tags it as -# rlinf-upstream-${BUILD_TARGET}, then this Dockerfile layers EFA on top. -# For local builds, run the upstream build first: +# This FROM is a placeholder -- kubernetes/libero/build-push.sh builds the +# upstream RLinf image first, tags it as rlinf-upstream-${BUILD_TARGET}, then +# builds this Dockerfile to layer EFA on top. To build the upstream image by +# hand first: # cd /path/to/RLinf -# docker build --build-arg BUILD_TARGET=embodied-maniskill_libero \ -# -t rlinf-upstream-embodied-maniskill_libero -f docker/Dockerfile . +# docker build --build-arg BUILD_TARGET=embodied-libero \ +# -t rlinf-upstream-embodied-libero -f docker/Dockerfile . # ---- Stage 2: EFA networking overlay ---- @@ -187,12 +187,13 @@ COPY *.patch /workspace/eks/patches/ # ============================================================================= # Copy source repos into the build context. # -# RLinf is always available (cloned in pre_build). RLinf is NOT pip-installed: -# the launchers run from /workspace/RLinf (cwd on sys.path), and the upstream -# image deliberately uses --no-install-project. DreamZero (groot package) is -# always cloned by the buildspec (pre_build) and made available on PYTHONPATH -# via DREAMZERO_PATH; its `dreamzero` venv is built by the upstream -# embodied-libero target (RLinf PR #1272), so this overlay does not rebuild it. +# RLinf is always available (cloned into the build context by build-push.sh +# before the build). RLinf is NOT pip-installed: the launchers run from +# /workspace/RLinf (cwd on sys.path), and the upstream image deliberately uses +# --no-install-project. DreamZero (groot package) is likewise cloned by +# build-push.sh and made available on PYTHONPATH via DREAMZERO_PATH; its +# `dreamzero` venv is built by the upstream embodied-libero target +# (RLinf PR #1272), so this overlay does not rebuild it. # ============================================================================= COPY RLinf/ /workspace/RLinf/ COPY DreamZero/ /workspace/DreamZero/ diff --git a/3.test_cases/pytorch/dreamzero/README.md b/3.test_cases/pytorch/dreamzero/README.md index 671dfb040..8db744394 100644 --- a/3.test_cases/pytorch/dreamzero/README.md +++ b/3.test_cases/pytorch/dreamzero/README.md @@ -57,7 +57,7 @@ each) — provisioned however you like (a static managed node group, a Capacity Block reservation, or on-demand autoscaling such as Karpenter; the workload is fixed-size, so autoscaling is a convenience, not a requirement) — plus EFA networking, the KubeRay operator, and FSx for Lustre shared storage. See -[`1.architectures/4.amazon-eks`](../../../1.architectures/4.amazon-eks) for +[`Amazon EKS distributed training architecture`](../../../1.architectures/4.amazon-eks/README.md) for cluster setup. The detailed prerequisite checklist lives in the walkthrough below. ## Full walkthrough diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-eval.yaml b/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-eval.yaml index 573a754b1..3ea04fda2 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-eval.yaml +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-eval.yaml @@ -19,7 +19,7 @@ # - umt5-xxl tokenizer + DreamZero-DROID backbone staged on FSx # (/fsx/models/umt5-xxl, /fsx/models/DreamZero-DROID) # - LIBERO dataset/assets staged (model-download.yaml) -# - Container image built and pushed to ECR (examples/buildspec.yml) +# - Container image built and pushed to ECR (build-push.sh) # - FSx PVC "fsx-claim" bound; training-sa ServiceAccount exists # # ConfigMaps (create BOTH before applying this manifest): From 2696d75e70e022eab9b6abe094c4aad0cb52ac5a Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Sat, 20 Jun 2026 15:08:58 -0500 Subject: [PATCH 23/45] docs(dreamzero): clarify eval intro is the pipeline run, not an accuracy result 'evaluate the result in the LIBERO simulator and render in-sim rollout videos' -> 'run the LIBERO simulator eval (which renders in-sim rollout videos)'. The eval + video path is shipped and validated end-to-end, but a 1-step checkpoint yields success_once=0.0 (documented in step 6); this avoids implying the intro promises a meaningful accuracy result. --- 3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md index 58abcee13..044355030 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md @@ -16,7 +16,7 @@ Kubernetes manifests that run on Amazon EKS: take the released 14B checkpoint (pretrained on DROID, a Franka arm), continue **supervised fine-tuning (SFT)** on the **LIBERO** benchmark (also a Franka arm, in simulation; `embodiment_tag: libero_sim`), then -**evaluate the result in the LIBERO simulator** and render in-sim rollout videos. +**run the LIBERO simulator eval** (which renders in-sim rollout videos). There is no native LIBERO 14B checkpoint upstream — warm-starting from DROID is the point. From 8403f656b785c54ba88617d25f750f441b6d3efe Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Sat, 20 Jun 2026 15:24:56 -0500 Subject: [PATCH 24/45] docs(dreamzero): reword warm-start rationale for clarity Replace 'There is no native LIBERO 14B checkpoint upstream -- warm-starting from DROID is the point' with customer-facing framing: the released DreamZero-DROID checkpoint is the 14B foundation weight you warm-start from, and continue-SFT adapts it to your target data (here LIBERO), the same pattern you'd follow with your own dataset. The old wording used insider framing and was imprecise (a 5B LIBERO checkpoint does exist upstream; only a 14B LIBERO one does not). --- 3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md index 044355030..88133ad85 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md @@ -17,8 +17,9 @@ Kubernetes manifests that run on Amazon EKS: take the released fine-tuning (SFT)** on the **LIBERO** benchmark (also a Franka arm, in simulation; `embodiment_tag: libero_sim`), then **run the LIBERO simulator eval** (which renders in-sim rollout videos). -There is no native LIBERO 14B checkpoint upstream — warm-starting from DROID is -the point. +The released DreamZero-DROID checkpoint is the 14B foundation weight you +warm-start from; continue-SFT adapts it to your target data (here, LIBERO) — +the same pattern you would follow with your own dataset. Upstream projects: [github.com/RLinf/RLinf](https://github.com/RLinf/RLinf) (training framework) and From 9b9c1f74271f0eb99a413668428e929148a14ee4 Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Sun, 21 Jun 2026 11:54:24 -0500 Subject: [PATCH 25/45] fix(dreamzero): run SFT launcher from ConfigMap mount, not deleted scripts dir The RayJob entrypoint copied the launcher to /workspace/eks/scripts/, but that directory no longer exists in the image: an earlier cleanup changed 'COPY docker/scripts/ /workspace/eks/scripts/' to 'COPY *.patch /workspace/eks/patches/', so the image now only creates /workspace/eks/patches. The cp failed with 'No such file or directory' (exit 127) and the RayJob never started. Run the launcher directly from its read-only ConfigMap mount at /tmp/scripts (it is path-independent -- it cd's to /workspace/RLinf itself), with a fallback to /workspace/eks/scripts for images that still bake it in. Caught by a live multi-step SFT run on 2x p5en. --- .../kubernetes/libero/dreamzero-sft.yaml | 15 +++++++++------ 1 file changed, 9 insertions(+), 6 deletions(-) diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-sft.yaml b/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-sft.yaml index a2f34e4a2..9186ac993 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-sft.yaml +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-sft.yaml @@ -44,15 +44,18 @@ metadata: reference: rlinf example: dreamzero-sft spec: - # Run the launcher on the head once the Ray cluster is ready. The launcher - # copies itself from the mounted ConfigMap (if present) then execs training. + # Run the launcher on the head once the Ray cluster is ready. The launcher is + # mounted read-only from the dreamzero-sft-launcher ConfigMap at /tmp/scripts; + # run it directly from there (it is path-independent -- it cd's to + # /workspace/RLinf itself). A fallback to a baked-in copy is kept for images + # that ship the launcher under /workspace/eks/scripts. entrypoint: >- bash -c ' if [ -f /tmp/scripts/run_dreamzero_sft_eks.sh ]; then - cp /tmp/scripts/run_dreamzero_sft_eks.sh /workspace/eks/scripts/run_dreamzero_sft_eks.sh; - chmod +x /workspace/eks/scripts/run_dreamzero_sft_eks.sh; - fi; - bash /workspace/eks/scripts/run_dreamzero_sft_eks.sh' + bash /tmp/scripts/run_dreamzero_sft_eks.sh; + else + bash /workspace/eks/scripts/run_dreamzero_sft_eks.sh; + fi' shutdownAfterJobFinishes: true ttlSecondsAfterFinished: 600 # The KubeRay submitter pod runs `ray job submit` against the head. It does NOT From 7808a38ba11a273b4b2589f4d8f12965bc9f75e9 Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Sun, 21 Jun 2026 12:44:06 -0500 Subject: [PATCH 26/45] docs(dreamzero): bullet the upstream-projects list in libero README --- 3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md index 88133ad85..f88fc5f05 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md @@ -22,8 +22,8 @@ warm-start from; continue-SFT adapts it to your target data (here, LIBERO) — the same pattern you would follow with your own dataset. Upstream projects: -[github.com/RLinf/RLinf](https://github.com/RLinf/RLinf) (training framework) and -[github.com/RLinf/dreamzero](https://github.com/RLinf/dreamzero) (the `groot` +- [github.com/RLinf/RLinf](https://github.com/RLinf/RLinf) (training framework) +- [github.com/RLinf/dreamzero](https://github.com/RLinf/dreamzero) (the `groot` package that provides the WAM model code). > **Validation scope (read this first).** The pipeline below was validated From 7945a0cf26aa3de18fc68af421b1e261ef122ced Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Sun, 21 Jun 2026 12:45:31 -0500 Subject: [PATCH 27/45] docs(dreamzero): add real 300-step training-convergence result Both READMEs' validation-scope sections now report the actual multi-step run on 2x p5en (FSDP2 + KubeRay, the shipped stack): train/loss 0.232 -> 0.085 over 300 steps (~6.9 s/step), a 207 GB DCP checkpoint written with zero UnpicklingError (exercising the gloo-coordinator fix at the full 16.48B scale). Framed honestly as 'trains and converges', NOT a task-accuracy claim (300 steps is short; the released 14B trained for 100K). The 1-step success_once=0.0 note and the upstream 5B ~96.7% accuracy reference are retained. --- 3.test_cases/pytorch/dreamzero/README.md | 28 +++++++++++++------ .../dreamzero/kubernetes/libero/README.md | 17 ++++++----- 2 files changed, 29 insertions(+), 16 deletions(-) diff --git a/3.test_cases/pytorch/dreamzero/README.md b/3.test_cases/pytorch/dreamzero/README.md index 8db744394..62e62e0aa 100644 --- a/3.test_cases/pytorch/dreamzero/README.md +++ b/3.test_cases/pytorch/dreamzero/README.md @@ -68,15 +68,25 @@ SFT RayJob → DCP→`.pt` conversion → LIBERO simulator eval. ## Results / validation status -The pipeline was validated **end-to-end** with a **1-step SFT smoke run** on 2× -`p5en.48xlarge`: the KubeRay RayJob reached `SUCCEEDED`, a **209 GB sharded FSDP -DCP checkpoint** was written, and it was converted to a single **91.7 GB `.pt`** -on CPU and consumed by the LIBERO simulator eval. This proves the *infrastructure -and pipeline* (image build, multi-node EFA/NCCL, FSDP2 sharded checkpointing, -DCP→`.pt` conversion, and in-sim eval) — **not** task accuracy. A 1-step -checkpoint yields `eval/success_once = 0.0`, which is **expected**; real accuracy -requires a multi-step SFT run (raise `runner.max_steps`). For reference, RLinf's -own LIBERO-Spatial SFT of the **5B** WAM +**Infrastructure & pipeline** — validated **end-to-end** on 2× `p5en.48xlarge`: +the KubeRay RayJob reached `SUCCEEDED`, a sharded FSDP2 DCP checkpoint was +written, converted to a single `.pt` on CPU, and consumed by the LIBERO +simulator eval (image build → multi-node EFA/NCCL → FSDP2 sharded checkpointing → +DCP→`.pt` conversion → in-sim eval). + +**Training convergence** — a **300-step** SFT run on 2× `p5en.48xlarge` (16× +H200, FSDP2 `full_shard` over EFA, ~6.9 s/step) reduced `train/loss` from +**0.232 → 0.085** with a clean, monotonic-ish curve (`action_loss` and +`dynamics_loss` both ~0.04–0.05 at the end). It wrote a **207 GB** DCP checkpoint +(16 shards + `.metadata`) with **zero** `UnpicklingError` — exercising the +in-image `dcp-save-gloo-coordinator.patch` at the full 16.48B-parameter scale. +This demonstrates the pipeline *trains and converges*, not a converged policy: +300 steps is a short run (the released 14B checkpoint trained for 100K), so it is +**not** a task-accuracy claim. + +**Accuracy reference** — a 1-step checkpoint yields `eval/success_once = 0.0` +(expected — it validates the eval machinery, not competence). For real accuracy, +train longer; RLinf's own LIBERO-Spatial SFT of the **5B** WAM ([`RLinf-DreamZero-WAN2.2-5B-LIBERO-SFT-Step18000`](https://huggingface.co/RLinf/RLinf-DreamZero-WAN2.2-5B-LIBERO-SFT-Step18000)) reaches **~96.7% `success_once` by step 18000** ([RLinf docs](https://rlinf.readthedocs.io/en/latest/rst_source/examples/embodied/sft_dreamzero.html)), confirming the recipe converges with sufficient steps. diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md index f88fc5f05..9ac3049d9 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md @@ -27,13 +27,16 @@ Upstream projects: package that provides the WAM model code). > **Validation scope (read this first).** The pipeline below was validated -> end-to-end on EKS with a **1-step** SFT run. That proves the *infrastructure -> and the pipeline* — image build, multi-node EFA/NCCL, FSDP2 sharded -> checkpointing, DCP→`.pt` conversion, and LIBERO simulator eval — **not** task -> accuracy. A 1-step checkpoint yields `eval/success_once = 0.0`, which is -> expected. Real accuracy requires a multi-step training run (the pipeline -> supports it — raise `runner.max_steps`; see step 4). No success numbers or loss -> curves are fabricated here. +> end-to-end on EKS: image build, multi-node EFA/NCCL, FSDP2 sharded +> checkpointing, DCP→`.pt` conversion, and LIBERO simulator eval. A **300-step** +> SFT run on 2× `p5en.48xlarge` reduced `train/loss` **0.232 → 0.085** (~6.9 +> s/step) and wrote a **207 GB** DCP checkpoint with **zero** `UnpicklingError`, +> confirming the gloo-coordinator checkpoint fix at the full 16.48B scale. That +> demonstrates the pipeline *trains and converges* — it is **not** a +> task-accuracy claim: 300 steps is a short run (the released 14B checkpoint +> trained for 100K). A 1-step checkpoint yields `eval/success_once = 0.0`, which +> is expected. For real accuracy, raise `runner.max_steps` (see step 4). No +> success numbers or loss curves are fabricated here. The recipe has six moving parts, each a self-contained manifest in this directory: From 26d5cc2a3e9b03717cb208b3bfdf60f066a696fb Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Sun, 21 Jun 2026 12:53:20 -0500 Subject: [PATCH 28/45] docs(dreamzero): de-jargon the checkpoint-fix mention in validation notes The validation-scope summaries (near the top of both READMEs) were the first place a reader met 'UnpicklingError' / 'gloo-coordinator fix', but the explanation only appears in the troubleshooting table (libero README) and nowhere in the root README. Reword to plain outcome language -- 'no corruption or save-time crashes' -- keeping the patch link (and a 'see Troubleshooting' pointer). UnpicklingError now appears only in the troubleshooting row, where a reader who hits it would look, with full context. Also made the root README's patch reference a clickable link. --- 3.test_cases/pytorch/dreamzero/README.md | 8 +++++--- .../pytorch/dreamzero/kubernetes/libero/README.md | 6 ++++-- 2 files changed, 9 insertions(+), 5 deletions(-) diff --git a/3.test_cases/pytorch/dreamzero/README.md b/3.test_cases/pytorch/dreamzero/README.md index 62e62e0aa..1d781cb69 100644 --- a/3.test_cases/pytorch/dreamzero/README.md +++ b/3.test_cases/pytorch/dreamzero/README.md @@ -77,9 +77,11 @@ DCP→`.pt` conversion → in-sim eval). **Training convergence** — a **300-step** SFT run on 2× `p5en.48xlarge` (16× H200, FSDP2 `full_shard` over EFA, ~6.9 s/step) reduced `train/loss` from **0.232 → 0.085** with a clean, monotonic-ish curve (`action_loss` and -`dynamics_loss` both ~0.04–0.05 at the end). It wrote a **207 GB** DCP checkpoint -(16 shards + `.metadata`) with **zero** `UnpicklingError` — exercising the -in-image `dcp-save-gloo-coordinator.patch` at the full 16.48B-parameter scale. +`dynamics_loss` both ~0.04–0.05 at the end). It wrote a **207 GB** sharded DCP +checkpoint (16 shards + `.metadata`) with no corruption or save-time crashes — +exercising the in-image checkpoint-save fix +([`dcp-save-gloo-coordinator.patch`](dcp-save-gloo-coordinator.patch)) at the +full 16.48B-parameter scale. This demonstrates the pipeline *trains and converges*, not a converged policy: 300 steps is a short run (the released 14B checkpoint trained for 100K), so it is **not** a task-accuracy claim. diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md index 9ac3049d9..9d00c054d 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md @@ -30,8 +30,10 @@ package that provides the WAM model code). > end-to-end on EKS: image build, multi-node EFA/NCCL, FSDP2 sharded > checkpointing, DCP→`.pt` conversion, and LIBERO simulator eval. A **300-step** > SFT run on 2× `p5en.48xlarge` reduced `train/loss` **0.232 → 0.085** (~6.9 -> s/step) and wrote a **207 GB** DCP checkpoint with **zero** `UnpicklingError`, -> confirming the gloo-coordinator checkpoint fix at the full 16.48B scale. That +> s/step) and wrote a **207 GB** sharded checkpoint with no corruption or +> save-time crashes — exercising the in-image checkpoint-save fix +> ([dcp-save-gloo-coordinator.patch](../../dcp-save-gloo-coordinator.patch); see +> Troubleshooting) at the full 16.48B scale. That > demonstrates the pipeline *trains and converges* — it is **not** a > task-accuracy claim: 300 steps is a short run (the released 14B checkpoint > trained for 100K). A 1-step checkpoint yields `eval/success_once = 0.0`, which From 82816d47ad9b90bd2fa9563d47c3173e43db8498 Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Sun, 21 Jun 2026 12:57:25 -0500 Subject: [PATCH 29/45] docs(dreamzero): link Troubleshooting + explain 14B vs 16.48B - libero README: make 'see Troubleshooting' a real anchor link (#troubleshooting). - Reconcile the 14B/16.48B discrepancy that appeared unexplained: '14B' is the Wan video-diffusion DiT backbone (headline); '16.48B' is the full trainable WAM once the action/state encoders + action head are added (live run reports 16,484,292,448 params). Added a callout box in the libero README and a concise inline gloss in the root README so the two figures are no longer ambiguous. --- 3.test_cases/pytorch/dreamzero/README.md | 4 +++- .../pytorch/dreamzero/kubernetes/libero/README.md | 9 ++++++++- 2 files changed, 11 insertions(+), 2 deletions(-) diff --git a/3.test_cases/pytorch/dreamzero/README.md b/3.test_cases/pytorch/dreamzero/README.md index 1d781cb69..8896c49d8 100644 --- a/3.test_cases/pytorch/dreamzero/README.md +++ b/3.test_cases/pytorch/dreamzero/README.md @@ -81,7 +81,9 @@ H200, FSDP2 `full_shard` over EFA, ~6.9 s/step) reduced `train/loss` from checkpoint (16 shards + `.metadata`) with no corruption or save-time crashes — exercising the in-image checkpoint-save fix ([`dcp-save-gloo-coordinator.patch`](dcp-save-gloo-coordinator.patch)) at the -full 16.48B-parameter scale. +full **16.48B**-parameter scale (the 14B Wan DiT backbone plus the action/state +encoders and action head — `16,484,292,448` trainable params; "14B" is the +backbone headline). This demonstrates the pipeline *trains and converges*, not a converged policy: 300 steps is a short run (the released 14B checkpoint trained for 100K), so it is **not** a task-accuracy claim. diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md index 9d00c054d..6effa2b78 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md @@ -10,6 +10,13 @@ matching. The model predicts both what will happen (video) and what to do (actions); the video prediction acts as a computational scaffold for action reasoning. +> **14B vs 16.48B.** "14B" is the Wan video-diffusion **DiT backbone** — the +> headline figure. The full trainable model is **16.48B** parameters once the +> action/state encoders and the action-head projector are added on top of the +> backbone (the live run reports `16,484,292,448`). This doc uses **14B** as the +> name and **16.48B** where the exact instantiated size matters (FSDP sharding, +> VRAM, checkpoint size). + This walkthrough packages the canonical customer workflow as a set of reusable Kubernetes manifests that run on Amazon EKS: take the released [`GEAR-Dreams/DreamZero-DROID`](https://huggingface.co/GEAR-Dreams/DreamZero-DROID) @@ -33,7 +40,7 @@ package that provides the WAM model code). > s/step) and wrote a **207 GB** sharded checkpoint with no corruption or > save-time crashes — exercising the in-image checkpoint-save fix > ([dcp-save-gloo-coordinator.patch](../../dcp-save-gloo-coordinator.patch); see -> Troubleshooting) at the full 16.48B scale. That +> [Troubleshooting](#troubleshooting)) at the full 16.48B scale. That > demonstrates the pipeline *trains and converges* — it is **not** a > task-accuracy claim: 300 steps is a short run (the released 14B checkpoint > trained for 100K). A 1-step checkpoint yields `eval/success_once = 0.0`, which From 9a7c63d2672b8015836d27c80ab4a9f8bbf1cb7c Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Sun, 21 Jun 2026 13:01:41 -0500 Subject: [PATCH 30/45] docs(dreamzero): reconcile 14B / 16.48B / 23B param figures precisely The HF model card publishes the checkpoint as '14B' (23B on-disk safetensors incl. frozen encoders); our live SFT run trains 16,484,292,448 params. Rewrite the callout to cover all three scopes of the SAME checkpoint and stop implying the encoders/head are added on top -- they are part of the published WAM: - 14B = publisher headline (Wan DiT backbone) - 16.48B = trainable params when instantiated for full SFT (backbone + the model's action/state encoders + action head) - 23B = on-disk total (also includes frozen CLIP / UMT5-XXL / Wan VAE) '14B checkpoint' phrasing is kept (it matches the publisher); root README gloss tightened to match and point at the callout. --- 3.test_cases/pytorch/dreamzero/README.md | 6 +++--- .../dreamzero/kubernetes/libero/README.md | 18 ++++++++++++------ 2 files changed, 15 insertions(+), 9 deletions(-) diff --git a/3.test_cases/pytorch/dreamzero/README.md b/3.test_cases/pytorch/dreamzero/README.md index 8896c49d8..2756573da 100644 --- a/3.test_cases/pytorch/dreamzero/README.md +++ b/3.test_cases/pytorch/dreamzero/README.md @@ -81,9 +81,9 @@ H200, FSDP2 `full_shard` over EFA, ~6.9 s/step) reduced `train/loss` from checkpoint (16 shards + `.metadata`) with no corruption or save-time crashes — exercising the in-image checkpoint-save fix ([`dcp-save-gloo-coordinator.patch`](dcp-save-gloo-coordinator.patch)) at the -full **16.48B**-parameter scale (the 14B Wan DiT backbone plus the action/state -encoders and action head — `16,484,292,448` trainable params; "14B" is the -backbone headline). +full **16.48B**-parameter scale (the released checkpoint is published as 14B — +its Wan DiT backbone — but trains `16,484,292,448` params once instantiated as a +full WAM; see the walkthrough's "14B vs 16.48B vs 23B" note). This demonstrates the pipeline *trains and converges*, not a converged policy: 300 steps is a short run (the released 14B checkpoint trained for 100K), so it is **not** a task-accuracy claim. diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md index 6effa2b78..c10def7df 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md @@ -10,12 +10,18 @@ matching. The model predicts both what will happen (video) and what to do (actions); the video prediction acts as a computational scaffold for action reasoning. -> **14B vs 16.48B.** "14B" is the Wan video-diffusion **DiT backbone** — the -> headline figure. The full trainable model is **16.48B** parameters once the -> action/state encoders and the action-head projector are added on top of the -> backbone (the live run reports `16,484,292,448`). This doc uses **14B** as the -> name and **16.48B** where the exact instantiated size matters (FSDP sharding, -> VRAM, checkpoint size). +> **14B vs 16.48B vs 23B.** These all describe the same released checkpoint, +> at different scopes: +> - **14B** — the publisher's headline (the Wan video-diffusion **DiT backbone**; +> the HF model card lists "14 Billion" and base model `Wan2.1-I2V-14B-480P`). +> - **16.48B** — the **trainable** parameters when the WAM is instantiated for +> full SFT: the DiT backbone together with the model's action/state encoders +> and action-head projector (the live run reports `16,484,292,448`). +> - **23B** — the **on-disk** safetensors total, which also includes the frozen +> conditioning encoders (CLIP / UMT5-XXL text / Wan VAE) that are not trained. +> +> This doc uses **14B** as the name and **16.48B** where the exact trainable size +> matters (FSDP sharding, VRAM, checkpoint size). This walkthrough packages the canonical customer workflow as a set of reusable Kubernetes manifests that run on Amazon EKS: take the released From 708172f63eb9d7c8ec48558298496544d39ca848 Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Sun, 21 Jun 2026 13:09:10 -0500 Subject: [PATCH 31/45] docs(dreamzero): add non-commercial model-license notice (legal) Per legal review: the GEAR-Dreams/DreamZero-DROID model is released under a non-commercial license (CC-BY-NC-4.0). Add a prominent notice at the top of both READMEs stating that any production use needs additional approvals and that users should review the license terms before using the model to make an informed decision. The notice scopes itself to the MODEL only; the repository code remains MIT-0 (unchanged). --- 3.test_cases/pytorch/dreamzero/README.md | 9 +++++++++ .../pytorch/dreamzero/kubernetes/libero/README.md | 9 +++++++++ 2 files changed, 18 insertions(+) diff --git a/3.test_cases/pytorch/dreamzero/README.md b/3.test_cases/pytorch/dreamzero/README.md index 2756573da..d603ea767 100644 --- a/3.test_cases/pytorch/dreamzero/README.md +++ b/3.test_cases/pytorch/dreamzero/README.md @@ -3,6 +3,15 @@ # DreamZero LIBERO 14B SFT (World-Action Model) on Amazon EKS +> **⚠️ Non-commercial model license.** This test case uses the +> [`GEAR-Dreams/DreamZero-DROID`](https://huggingface.co/GEAR-Dreams/DreamZero-DROID) +> model, which is released under a **non-commercial license** +> ([CC-BY-NC-4.0](https://spdx.org/licenses/CC-BY-NC-4.0)). Any production use of +> this model needs additional approvals. Note the license terms before using this +> model so you can make an informed decision about whether to use it. (This note +> concerns the *model* only; the code in this repository is MIT-0 — see +> [License](#license).) + This test case is a working example of **deploying DreamZero training on Amazon EKS**. DreamZero is a **14B-parameter World-Action Model (WAM)** — a Wan-based, causal (autoregressive) video diffusion transformer that *jointly* denoises diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md index c10def7df..4dda65828 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md @@ -3,6 +3,15 @@ # Continue SFT of DreamZero (14B World-Action Model) on LIBERO on Amazon EKS +> **⚠️ Non-commercial model license.** This walkthrough uses the +> [`GEAR-Dreams/DreamZero-DROID`](https://huggingface.co/GEAR-Dreams/DreamZero-DROID) +> model, which is released under a **non-commercial license** +> ([CC-BY-NC-4.0](https://spdx.org/licenses/CC-BY-NC-4.0)). Any production use of +> this model needs additional approvals. Note the license terms before using this +> model so you can make an informed decision about whether to use it. (This note +> concerns the *model* only; the code in this repository is MIT-0 — see +> [License](#license).) + DreamZero is a **14B-parameter World-Action Model (WAM)** — a Wan-based, causal (autoregressive) video-diffusion Diffusion Transformer (DiT) that *jointly* denoises future video frames and future robot actions via flow From 6e34e8e4b3f61746e5789d4bad1d05d15306d373 Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Sun, 21 Jun 2026 13:15:27 -0500 Subject: [PATCH 32/45] docs(dreamzero): list optional FSX_* vars in env_vars.example The storage/ manifests use FSX_SUBNET_ID, FSX_SECURITY_GROUP_IDS, FSX_FILESYSTEM_ID, FSX_DNS_NAME, and FSX_MOUNT_NAME, but env_vars.example only listed the always-needed vars. Add the FSX_* vars (commented out, marked optional -- only for customers who provision FSx via storage/*.yaml rather than reusing an existing fsx-claim). The README already gives these as inline exports at the storage step; this just makes the env-var reference complete. No manifest or behavior change. --- .../dreamzero/kubernetes/libero/env_vars.example | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/env_vars.example b/3.test_cases/pytorch/dreamzero/kubernetes/libero/env_vars.example index 6d6bebc7b..d536d0e86 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/env_vars.example +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/env_vars.example @@ -16,3 +16,16 @@ export AWS_REGION=us-east-1 # Pinned source refs (baked into the image build). export UPSTREAM_REF=b3bbabb1f461 export DREAMZERO_REF=ab790c198fbc + +# --------------------------------------------------------------------------- +# OPTIONAL: only needed if you provision FSx via the storage/ manifests. +# Skip these if you already have an `fsx-claim` PVC bound in your namespace. +# +# For storage/pvc-fsx-lustre-dynamic.yaml (provision a NEW filesystem): +# export FSX_SUBNET_ID=subnet-0abc... +# export FSX_SECURITY_GROUP_IDS=sg-0def... +# +# For storage/pv-fsx-lustre-static.yaml (bind an EXISTING filesystem): +# export FSX_FILESYSTEM_ID=fs-0... +# export FSX_DNS_NAME=fs-0....fsx..amazonaws.com +# export FSX_MOUNT_NAME=abcd1234 From 2076a60cab1e9395bca20ad874b2b21e1f196ea5 Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Sun, 21 Jun 2026 13:25:54 -0500 Subject: [PATCH 33/45] docs(dreamzero): fix comment accuracy + remove personal namespace from examples Comment audit before PR: - Remove personal namespace leak: 4 manifests had 'export NAMESPACE=natharno' and dreamzero-sft.yaml had 'rlinf' in their usage examples. Normalize all to 'dreamzero' (matching env_vars.example). - run_dreamzero_sft_eks.sh: fix the FSx free-space figure (>=200GB -> >=250GB, consistent with the READMEs and manifests); consolidate two overlapping save_full_model_weights/DCP comment blocks into one accurate block (the first vaguely said the gather is 'slow/stalls'; the accurate cause is the NCCL allgather_into_tensor_coalesced error); tighten the metadata comment. Net -8 lines, no behavior change. --- .../kubernetes/libero/convert-checkpoint.yaml | 2 +- .../kubernetes/libero/dreamzero-eval.yaml | 2 +- .../kubernetes/libero/dreamzero-sft.yaml | 2 +- .../kubernetes/libero/generate-metadata.yaml | 2 +- .../kubernetes/libero/model-download.yaml | 2 +- .../libero/scripts/run_dreamzero_sft_eks.sh | 36 ++++++++----------- 6 files changed, 19 insertions(+), 27 deletions(-) diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/convert-checkpoint.yaml b/3.test_cases/pytorch/dreamzero/kubernetes/libero/convert-checkpoint.yaml index 6c066f956..36b008715 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/convert-checkpoint.yaml +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/convert-checkpoint.yaml @@ -16,7 +16,7 @@ # # Usage (restricted envsubst protects the inline ${...} shell vars): # export ECR_URI=.dkr.ecr..amazonaws.com/ -# export NAMESPACE=natharno +# export NAMESPACE=dreamzero # # Optionally override STEP (default global_step_1): # kubectl -n $NAMESPACE create configmap dreamzero-convert-launcher \ # --from-file=convert_checkpoint.sh=kubernetes/libero/scripts/convert_checkpoint.sh \ diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-eval.yaml b/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-eval.yaml index 3ea04fda2..6ba80f1be 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-eval.yaml +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-eval.yaml @@ -36,7 +36,7 @@ # Apply (RESTRICTED envsubst -- substitute ONLY ${ECR_URI} and ${NAMESPACE} so the # inline ${PYTHONPATH:-} guards in the bootstrap are NOT clobbered): # export ECR_URI=.dkr.ecr..amazonaws.com/ -# export NAMESPACE=natharno +# export NAMESPACE=dreamzero # envsubst '${ECR_URI} ${NAMESPACE}' < kubernetes/libero/dreamzero-eval.yaml | kubectl apply -f - --- apiVersion: batch/v1 diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-sft.yaml b/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-sft.yaml index 9186ac993..ae7e1eff7 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-sft.yaml +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-sft.yaml @@ -28,7 +28,7 @@ # --from-file=run_dreamzero_sft_eks.sh=kubernetes/libero/scripts/run_dreamzero_sft_eks.sh \ # --dry-run=client -o yaml | kubectl apply -f - # export ECR_URI=.dkr.ecr..amazonaws.com/ -# export NAMESPACE=rlinf +# export NAMESPACE=dreamzero # # RESTRICTED envsubst -- substitute ONLY ${ECR_URI} and ${NAMESPACE}. # # (The Ray head-election bash is gone, but keep this habit: other inline # # shell snippets, e.g. the launcher copy, must not be expanded.) diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/generate-metadata.yaml b/3.test_cases/pytorch/dreamzero/kubernetes/libero/generate-metadata.yaml index 6426eb3a9..e17c90785 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/generate-metadata.yaml +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/generate-metadata.yaml @@ -32,7 +32,7 @@ # # Usage: # export ECR_URI=.dkr.ecr..amazonaws.com/ -# export NAMESPACE=natharno +# export NAMESPACE=dreamzero # # NOTE: use RESTRICTED envsubst -- substitute ONLY ${ECR_URI} and ${NAMESPACE}. # # The container command contains shell vars (${PYTHONPATH}, ${DREAMZERO_PATH}) # # that unrestricted envsubst would clobber to empty strings. Same pattern as diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/model-download.yaml b/3.test_cases/pytorch/dreamzero/kubernetes/libero/model-download.yaml index dcd798e38..e98dc87a5 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/model-download.yaml +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/model-download.yaml @@ -37,7 +37,7 @@ # LIBERO dataset (~152GB total). Ensure FSx has headroom. # # Usage: -# export NAMESPACE=natharno # shared test cluster namespace +# export NAMESPACE=dreamzero # envsubst < kubernetes/libero/model-download.yaml | kubectl apply -f - # kubectl logs -f job/model-download-dreamzero -n $NAMESPACE apiVersion: batch/v1 diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/run_dreamzero_sft_eks.sh b/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/run_dreamzero_sft_eks.sh index 36f649316..8b9583465 100755 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/run_dreamzero_sft_eks.sh +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/scripts/run_dreamzero_sft_eks.sh @@ -102,16 +102,12 @@ HYDRA_ARGS="${HYDRA_ARGS} data.train_data_paths=${DATASET_PATH}" HYDRA_ARGS="${HYDRA_ARGS} cluster.num_nodes=${NUM_NODES}" HYDRA_ARGS="${HYDRA_ARGS} runner.logger.log_path=${LOG_DIR}" -# NOTE on checkpoint format for eval: -# RLinf's FSDP saver writes a sharded DCP checkpoint (.distcp + .metadata) under -# {ckpt}/actor/dcp_checkpoint/. The LIBERO eval needs a single .pt; convert the -# DCP to .pt offline (CPU) with kubernetes/libero/convert-checkpoint.yaml. -# We do NOT use +actor.fsdp_config.save_full_model_weights=true: on 2x p5en the -# rank-0 full-state-dict gather for the 16B model is pathologically slow / stalls. -# DCP-only save + offline convert is faster and more reliable. -# IMPORTANT: ensure FSx has ample free space (>=200GB for a 14B DCP checkpoint); -# a full filesystem truncates torch.save mid-write (inline_container.cc -# "unexpected pos") producing corrupt, unreadable shards. +# Checkpoint format: RLinf's FSDP saver writes a sharded DCP checkpoint +# (.distcp + .metadata) under {ckpt}/actor/dcp_checkpoint/. The LIBERO eval needs +# a single .pt; convert it offline on CPU with convert-checkpoint.yaml. +# Ensure FSx has >=250GB free (a 14B DCP checkpoint is ~140-206GB): a full +# filesystem truncates torch.save mid-write (inline_container.cc "unexpected pos") +# producing corrupt, unreadable shards. # Temporal alignment fix: libero_sft_dreamzero_14b.yaml sets action_horizon=16 # but inherits num_action_per_block=24 from model/dreamzero_14b.yaml (a DROID @@ -125,20 +121,16 @@ HYDRA_ARGS="${HYDRA_ARGS} actor.model.num_action_per_block=16" # omits fsdp_config.save_full_model_weights, so it falls through to the code default # of True (rlinf/hybrid_engines/fsdp/fsdp_model_manager.py). On the 16B model that # triggers a full-state-dict gather -> "Backend nccl does not support -# allgather_into_tensor_coalesced" and (per upstream) a rank-0 gather that stalls. -# The sharded DCP checkpoint is the supported path; convert it to a single .pt -# offline on CPU (convert-checkpoint.yaml). The key is absent from the config struct, -# so it must be ADDED with the '+' prefix. +# allgather_into_tensor_coalesced". DCP-only save + offline convert is the +# supported path. The key is absent from the config struct, so ADD it with '+'. HYDRA_ARGS="${HYDRA_ARGS} +actor.fsdp_config.save_full_model_weights=false" -# Metadata handling: set metadata_json_path when METADATA_PATH is non-empty. -# It now DEFAULTS to /fsx/models/metadata-libero.json (LIBERO libero_sim stats -# generated by generate-metadata.yaml), so the override is passed by default. -# The DreamZero-DROID checkpoint only bundles oxe_droid metadata, so LIBERO SFT -# would fail with a KeyError on embodiment_tag 'libero_sim' without this. Set -# METADATA_PATH="" to fall back to the checkpoint's bundled metadata instead. -# NOTE: metadata_json_path is COMMENTED OUT in the upstream config (not in the -# Hydra struct), so it must be ADDED with the '+' prefix, not overridden. +# Metadata: the DreamZero-DROID checkpoint only bundles oxe_droid stats, so +# LIBERO SFT fails with a KeyError on embodiment_tag 'libero_sim' without +# libero_sim normalization metadata (generated by generate-metadata.yaml). +# metadata_json_path is commented out in the upstream config (not in the Hydra +# struct), so ADD it with '+'. Set METADATA_PATH="" to fall back to the +# checkpoint's bundled metadata instead. if [ -n "${METADATA_PATH}" ]; then HYDRA_ARGS="${HYDRA_ARGS} +actor.model.metadata_json_path=${METADATA_PATH}" fi From b557e551f179c101b121738187d69dce959bfef5 Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Sun, 21 Jun 2026 13:30:47 -0500 Subject: [PATCH 34/45] docs(dreamzero): note why static PVC uses storageClassName: "" Add a comment explaining the empty-string storageClassName on the static FSx PVC is intentional (disables dynamic provisioning so the PVC binds the named static PV, rather than the cluster default StorageClass provisioning a new volume). Prevents a future reader from 'fixing' it by adding a class, which would break the static bind. --- .../kubernetes/libero/storage/pv-fsx-lustre-static.yaml | 3 +++ 1 file changed, 3 insertions(+) diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/storage/pv-fsx-lustre-static.yaml b/3.test_cases/pytorch/dreamzero/kubernetes/libero/storage/pv-fsx-lustre-static.yaml index ff53f6d74..09d994aa2 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/storage/pv-fsx-lustre-static.yaml +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/storage/pv-fsx-lustre-static.yaml @@ -32,6 +32,9 @@ metadata: spec: accessModes: - ReadWriteMany + # storageClassName "" (empty, not unset) is intentional: it disables dynamic + # provisioning so the PVC binds the static PV named below instead of letting + # the cluster's default StorageClass provision a new volume. storageClassName: "" volumeName: fsx-pv-dreamzero resources: From 1f7ac97a408cf40b63bd67761d99365a53087c12 Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Sun, 21 Jun 2026 13:36:30 -0500 Subject: [PATCH 35/45] docs(dreamzero): clarify provisioner vs capacity-backing in Prerequisites The previous wording listed 'managed node group, Capacity Block reservation, or Karpenter' as parallel options, conflating two independent axes. Reword to: provisioner (static managed node group OR Karpenter) backed by a capacity reservation (Capacity Block for ML OR ODCR) -- either capacity type works with either provisioner. Drop 'on-demand' as a backing option since 2x p5en on-demand is effectively unobtainable; these instances come from a reservation. --- 3.test_cases/pytorch/dreamzero/README.md | 10 ++++++---- 1 file changed, 6 insertions(+), 4 deletions(-) diff --git a/3.test_cases/pytorch/dreamzero/README.md b/3.test_cases/pytorch/dreamzero/README.md index d603ea767..e88d9cba0 100644 --- a/3.test_cases/pytorch/dreamzero/README.md +++ b/3.test_cases/pytorch/dreamzero/README.md @@ -62,10 +62,12 @@ node**, and **FSx for Lustre with ≥250 GB free** (a 14B FSDP DCP checkpoint is ## Prerequisites An Amazon EKS cluster that can schedule **2× `p5en.48xlarge`** (8× H200 + EFA -each) — provisioned however you like (a static managed node group, a Capacity -Block reservation, or on-demand autoscaling such as Karpenter; the workload is -fixed-size, so autoscaling is a convenience, not a requirement) — plus EFA -networking, the KubeRay operator, and FSx for Lustre shared storage. See +each) — provisioned however you like (a static managed node group or Karpenter +for autoscaling, backed by a **Capacity Block for ML** or an **On-Demand +Capacity Reservation (ODCR)** — either capacity type works with either +provisioner; the workload is fixed-size, so autoscaling is a convenience, not a +requirement) — plus EFA networking, the KubeRay operator, and FSx for Lustre +shared storage. See [`Amazon EKS distributed training architecture`](../../../1.architectures/4.amazon-eks/README.md) for cluster setup. The detailed prerequisite checklist lives in the walkthrough below. From 1f638b5488acbf767328263c05599edd05fa88fa Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Sun, 21 Jun 2026 13:39:03 -0500 Subject: [PATCH 36/45] docs(dreamzero): reword loss-curve description 'clean, monotonic-ish curve' -> 'steady downward trend' -- more professional and accurate (the curve declined overall but had step-to-step noise, not strict monotonicity). --- 3.test_cases/pytorch/dreamzero/README.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/3.test_cases/pytorch/dreamzero/README.md b/3.test_cases/pytorch/dreamzero/README.md index e88d9cba0..726094a56 100644 --- a/3.test_cases/pytorch/dreamzero/README.md +++ b/3.test_cases/pytorch/dreamzero/README.md @@ -87,7 +87,7 @@ DCP→`.pt` conversion → in-sim eval). **Training convergence** — a **300-step** SFT run on 2× `p5en.48xlarge` (16× H200, FSDP2 `full_shard` over EFA, ~6.9 s/step) reduced `train/loss` from -**0.232 → 0.085** with a clean, monotonic-ish curve (`action_loss` and +**0.232 → 0.085** with a steady downward trend (`action_loss` and `dynamics_loss` both ~0.04–0.05 at the end). It wrote a **207 GB** sharded DCP checkpoint (16 shards + `.metadata`) with no corruption or save-time crashes — exercising the in-image checkpoint-save fix From d188598e358358c46c1fcbb065beec8d21884f51 Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Sun, 21 Jun 2026 13:42:28 -0500 Subject: [PATCH 37/45] docs(dreamzero): deep-link the root README to the 14B-vs-16.48B-vs-23B note The note is a blockquote callout (no auto-generated anchor). Add an explicit anchor before it in the libero README and turn the root README's plain-text reference into a deep link (kubernetes/libero/README.md#param-counts) that jumps straight to it. --- 3.test_cases/pytorch/dreamzero/README.md | 2 +- 3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md | 2 ++ 2 files changed, 3 insertions(+), 1 deletion(-) diff --git a/3.test_cases/pytorch/dreamzero/README.md b/3.test_cases/pytorch/dreamzero/README.md index 726094a56..679042030 100644 --- a/3.test_cases/pytorch/dreamzero/README.md +++ b/3.test_cases/pytorch/dreamzero/README.md @@ -94,7 +94,7 @@ exercising the in-image checkpoint-save fix ([`dcp-save-gloo-coordinator.patch`](dcp-save-gloo-coordinator.patch)) at the full **16.48B**-parameter scale (the released checkpoint is published as 14B — its Wan DiT backbone — but trains `16,484,292,448` params once instantiated as a -full WAM; see the walkthrough's "14B vs 16.48B vs 23B" note). +full WAM; see the walkthrough's [**14B vs 16.48B vs 23B**](kubernetes/libero/README.md#param-counts) note). This demonstrates the pipeline *trains and converges*, not a converged policy: 300 steps is a short run (the released 14B checkpoint trained for 100K), so it is **not** a task-accuracy claim. diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md index 4dda65828..494008d68 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md @@ -19,6 +19,8 @@ matching. The model predicts both what will happen (video) and what to do (actions); the video prediction acts as a computational scaffold for action reasoning. + + > **14B vs 16.48B vs 23B.** These all describe the same released checkpoint, > at different scopes: > - **14B** — the publisher's headline (the Wan video-diffusion **DiT backbone**; From 33c7184d3b9686bc249b42b1afd3dfbaabafd90d Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Sun, 21 Jun 2026 13:59:35 -0500 Subject: [PATCH 38/45] fix(dreamzero): drop no-op network-layer topology constraint + false latency claim The topologySpreadConstraints used topologyKey topology.k8s.aws/network-node-layer-2, a label applied only by SageMaker HyperPod -- vanilla EKS nodes (incl. this cluster) do not carry it, so with whenUnsatisfiable: ScheduleAnyway the constraint was a no-op. It was also semantically backwards: a spread constraint spreads pods across domains, while the README claimed it 'prefers co-location ... for lowest NCCL latency' (co-location would need podAffinity, and the inter-node network proximity it targets is a HyperPod-only capability here). Remove the dead constraint from the head + worker (keeping the working podAntiAffinity that pins one pod per node via kubernetes.io/hostname) and drop the inaccurate README sentence. Soft scheduling change only; training correctness unaffected. --- .../pytorch/dreamzero/kubernetes/libero/README.md | 3 +-- .../dreamzero/kubernetes/libero/dreamzero-sft.yaml | 14 -------------- 2 files changed, 1 insertion(+), 16 deletions(-) diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md index 494008d68..652fa21b5 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md @@ -97,8 +97,7 @@ scheduler fans **FSDP2 `full_shard`** training across all 16 GPUs (the 16.48B model is *sharded*, not replicated). There is **no** torchrun, DeepSpeed, or manual head election. Gradient sync flows over **NCCL on EFA RDMA** (libfabric 2.4 / aws-ofi-nccl 1.18, GPUDirect RDMA). Pod anti-affinity guarantees one pod -per physical node; a topology-spread constraint prefers co-location under the -same network layer for lowest NCCL latency. `shutdownAfterJobFinishes: true` +per physical node. `shutdownAfterJobFinishes: true` tears the RayCluster down when training ends. ### The World-Action Model diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-sft.yaml b/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-sft.yaml index ae7e1eff7..03ad2a189 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-sft.yaml +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-sft.yaml @@ -100,13 +100,6 @@ spec: values: - dreamzero-sft topologyKey: kubernetes.io/hostname - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.k8s.aws/network-node-layer-2 - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - app: dreamzero-sft containers: - name: ray-head image: ${ECR_URI}:latest @@ -212,13 +205,6 @@ spec: values: - dreamzero-sft topologyKey: kubernetes.io/hostname - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.k8s.aws/network-node-layer-2 - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - app: dreamzero-sft containers: - name: ray-worker image: ${ECR_URI}:latest From 7a76f11e4249d5d5ad98f65660bae9d115c049be Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Sun, 21 Jun 2026 14:13:11 -0500 Subject: [PATCH 39/45] docs(dreamzero): spell out acronyms on first use (AWS convention) Expand technical acronyms at first body use (inline, AWS docs/blog style), skipping well-known ones (GPU/CPU/CUDA/NVIDIA) and title occurrences: - Root README: SFT, FSDP, EFA, NCCL, DCP, ECR, DROID. - libero README: FSDP, VRAM, EFA, NCCL, DCP, RDMA, CSI, VPC, IRSA, PVC, VAE, I2V (WAM/DiT/SFT were already spelled out). --- 3.test_cases/pytorch/dreamzero/README.md | 15 +++++++----- .../dreamzero/kubernetes/libero/README.md | 24 +++++++++++-------- 2 files changed, 23 insertions(+), 16 deletions(-) diff --git a/3.test_cases/pytorch/dreamzero/README.md b/3.test_cases/pytorch/dreamzero/README.md index 679042030..5af449931 100644 --- a/3.test_cases/pytorch/dreamzero/README.md +++ b/3.test_cases/pytorch/dreamzero/README.md @@ -16,14 +16,17 @@ This test case is a working example of **deploying DreamZero training on Amazon EKS**. DreamZero is a **14B-parameter World-Action Model (WAM)** — a Wan-based, causal (autoregressive) video diffusion transformer that *jointly* denoises future video frames and robot actions via flow matching. Here it continues -supervised fine-tuning from the released +supervised fine-tuning (SFT) from the released [`GEAR-Dreams/DreamZero-DROID`](https://huggingface.co/GEAR-Dreams/DreamZero-DROID) 14B checkpoint on the **LIBERO** manipulation benchmark and evaluates it in the -LIBERO simulator — exercising multi-node **FSDP2** training over **EFA** with +LIBERO simulator — exercising multi-node **FSDP2** (Fully Sharded Data Parallel) +training over **EFA** (Elastic Fabric Adapter) with **KubeRay**. The focus is the **EKS deployment mechanics** (image build, -multi-node EFA/NCCL, FSDP2 sharded checkpointing, sim eval), not a training-quality +multi-node EFA + NCCL (NVIDIA Collective Communications Library), FSDP2 sharded +checkpointing, sim eval), not a training-quality or transfer result. Note that LIBERO is a **simulation** of the same Franka Emika -Panda arm that DROID captures in the **real world**, so warm-starting the DROID +Panda arm that the Distributed Robot Interaction Dataset (DROID) captures in the +**real world**, so warm-starting the DROID checkpoint onto LIBERO must bridge a real→sim visual domain gap. LIBERO is used here as a convenient public dataset + simulator to exercise the pipeline end-to-end; in practice you would swap it for your own dataset (real or simulated) @@ -48,7 +51,7 @@ walkthrough of the World-Action Model — see ``` dreamzero/ ├── Dockerfile # two-stage RLinf + EFA overlay image -├── dcp-save-gloo-coordinator.patch # DCP checkpoint fix, applied to RLinf at build time +├── dcp-save-gloo-coordinator.patch # Distributed Checkpoint (DCP) save fix, applied to RLinf at build time ├── diagrams/ # WAM + infra topology (draw.io + SVG) └── kubernetes/libero/ # the EKS recipe (RayJob SFT + eval) — see its README ``` @@ -74,7 +77,7 @@ cluster setup. The detailed prerequisite checklist lives in the walkthrough belo ## Full walkthrough **Full step-by-step walkthrough: [`kubernetes/libero/README.md`](kubernetes/libero/README.md)** — -build-image → push-to-ECR → stage models/dataset → generate metadata → multi-node +build-image → push-to-ECR (Amazon Elastic Container Registry) → stage models/dataset → generate metadata → multi-node SFT RayJob → DCP→`.pt` conversion → LIBERO simulator eval. ## Results / validation status diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md index 652fa21b5..4f5acc88a 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md @@ -24,15 +24,17 @@ prediction acts as a computational scaffold for action reasoning. > **14B vs 16.48B vs 23B.** These all describe the same released checkpoint, > at different scopes: > - **14B** — the publisher's headline (the Wan video-diffusion **DiT backbone**; -> the HF model card lists "14 Billion" and base model `Wan2.1-I2V-14B-480P`). +> the HF model card lists "14 Billion" and base model `Wan2.1-I2V-14B-480P`, +> an image-to-video (I2V) model). > - **16.48B** — the **trainable** parameters when the WAM is instantiated for > full SFT: the DiT backbone together with the model's action/state encoders > and action-head projector (the live run reports `16,484,292,448`). > - **23B** — the **on-disk** safetensors total, which also includes the frozen -> conditioning encoders (CLIP / UMT5-XXL text / Wan VAE) that are not trained. +> conditioning encoders (CLIP image encoder / UMT5-XXL text encoder / Wan +> variational autoencoder (VAE)) that are not trained. > > This doc uses **14B** as the name and **16.48B** where the exact trainable size -> matters (FSDP sharding, VRAM, checkpoint size). +> matters (FSDP (Fully Sharded Data Parallel) sharding, VRAM (video RAM), checkpoint size). This walkthrough packages the canonical customer workflow as a set of reusable Kubernetes manifests that run on Amazon EKS: take the released @@ -51,8 +53,9 @@ Upstream projects: package that provides the WAM model code). > **Validation scope (read this first).** The pipeline below was validated -> end-to-end on EKS: image build, multi-node EFA/NCCL, FSDP2 sharded -> checkpointing, DCP→`.pt` conversion, and LIBERO simulator eval. A **300-step** +> end-to-end on EKS: image build, multi-node EFA (Elastic Fabric Adapter) + +> NCCL (NVIDIA Collective Communications Library), FSDP2 sharded +> checkpointing, Distributed Checkpoint (DCP)→`.pt` conversion, and LIBERO simulator eval. A **300-step** > SFT run on 2× `p5en.48xlarge` reduced `train/loss` **0.232 → 0.085** (~6.9 > s/step) and wrote a **207 GB** sharded checkpoint with no corruption or > save-time crashes — exercising the in-image checkpoint-save fix @@ -95,7 +98,8 @@ operator brings the Ray cluster up, then runs the Ray-agnostic launcher (`run_dreamzero_sft_eks.sh`) as the head `entrypoint`; RLinf's `Cluster` (Ray) scheduler fans **FSDP2 `full_shard`** training across all 16 GPUs (the 16.48B model is *sharded*, not replicated). There is **no** torchrun, DeepSpeed, or -manual head election. Gradient sync flows over **NCCL on EFA RDMA** (libfabric +manual head election. Gradient sync flows over **NCCL on EFA RDMA (Remote Direct +Memory Access)** (libfabric 2.4 / aws-ofi-nccl 1.18, GPUDirect RDMA). Pod anti-affinity guarantees one pod per physical node. `shutdownAfterJobFinishes: true` tears the RayCluster down when training ends. @@ -171,7 +175,7 @@ kubectl get pods -n kuberay-system ### 3. FSx for Lustre shared storage (`fsx-claim` at `/fsx`) -Every step mounts a `PersistentVolumeClaim` named **`fsx-claim`** at `/fsx`. If +Every step mounts a `PersistentVolumeClaim` (PVC) named **`fsx-claim`** at `/fsx`. If your cluster already exposes such a PVC (many EKS GPU cluster templates ship one), confirm it is `Bound` with `ReadWriteMany` access and has ≥250 GB free: @@ -182,11 +186,11 @@ kubectl get pvc fsx-claim -n "$NAMESPACE" ``` If you do **not** already have one, this directory ships two optional manifests -(both require the `fsx.csi.aws.com` CSI driver): +(both require the `fsx.csi.aws.com` CSI (Container Storage Interface) driver): - **Dynamic provisioning** — `storage/pvc-fsx-lustre-dynamic.yaml` creates a `StorageClass` + `fsx-claim` PVC and provisions a fresh filesystem. Fill in - `FSX_SUBNET_ID` and `FSX_SECURITY_GROUP_IDS` for your cluster's VPC: + `FSX_SUBNET_ID` and `FSX_SECURITY_GROUP_IDS` for your cluster's VPC (Virtual Private Cloud): ```bash export FSX_SUBNET_ID=subnet-0abc... FSX_SECURITY_GROUP_IDS=sg-0def... @@ -205,7 +209,7 @@ If you do **not** already have one, this directory ships two optional manifests ### 4. A `training-sa` ServiceAccount with credentials for HF + S3 The SFT, eval, convert, and metadata pods run as the **`training-sa`** -ServiceAccount. Give it IRSA or EKS Pod Identity so pods can reach Hugging Face +ServiceAccount. Give it IRSA (IAM Roles for Service Accounts) or EKS Pod Identity so pods can reach Hugging Face and (if you stage to/from S3) Amazon S3. For example, with IRSA via `eksctl`: ```bash From 3dd8de16321872707aeb09cd4f3efc1d99e7bd49 Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Sun, 21 Jun 2026 14:16:15 -0500 Subject: [PATCH 40/45] docs(dreamzero): fix libero prereq to not require autoscaling/Karpenter Prereq #1 still framed 'GPU autoscaling (e.g. Karpenter)' as a requirement, contradicting the root README fix (commit 1f7ac97a). The workload is fixed-size, so autoscaling is a convenience, not a requirement. Reword heading ('provision' -> 'schedule') and body to match: nodes can come from a static managed node group or Karpenter, backed by a Capacity Block for ML or an ODCR. --- .../pytorch/dreamzero/kubernetes/libero/README.md | 15 +++++++++------ 1 file changed, 9 insertions(+), 6 deletions(-) diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md index 4f5acc88a..b211f6383 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md @@ -139,12 +139,15 @@ only the action channel. ## Prerequisites -### 1. An EKS cluster that can provision 2× `p5en.48xlarge` with EFA - -You need an Amazon EKS cluster with GPU autoscaling (e.g. Karpenter) able to -launch **2× `p5en.48xlarge`** nodes, each with **8× H200** GPUs and **16 EFA -NICs**, plus the NVIDIA GPU Operator (or device plugin) and EFA device plugin so -pods can request `nvidia.com/gpu` and `vpc.amazonaws.com/efa`. Cluster-creation +### 1. An EKS cluster that can schedule 2× `p5en.48xlarge` with EFA + +You need an Amazon EKS cluster that can schedule **2× `p5en.48xlarge`** nodes, +each with **8× H200** GPUs and **16 EFA NICs** — provisioned however you like (a +static managed node group or Karpenter for autoscaling, backed by a Capacity +Block for ML or an On-Demand Capacity Reservation (ODCR); the workload is +fixed-size, so autoscaling is a convenience, not a requirement). You also need +the NVIDIA GPU Operator (or device plugin) and EFA device plugin so pods can +request `nvidia.com/gpu` and `vpc.amazonaws.com/efa`. Cluster-creation references live in [`1.architectures/4.amazon-eks`](../../../../../1.architectures/4.amazon-eks). From 0ae16dc5c2dff9452d497aa6576eec5b76c88f71 Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Sun, 21 Jun 2026 14:22:28 -0500 Subject: [PATCH 41/45] fix(dreamzero): recommend HF_TOKEN for rate limits + actually wire it in 'Hugging Face auth is optional' was accurate for access but glossed over a real reliability issue: anonymous HF downloads are rate-limited per source IP and the anonymous tier is much stricter than authenticated (HF's #1 cause of 429s). This download is large/multi-file and egresses through a shared EKS NAT gateway, so the cluster's other workloads share the same per-IP anonymous quota. - README: reword the auth note to 'optional, but recommended' and explain the per-IP rate-limit / shared-NAT contention; reword the download step to match. - model-download.yaml: the download script reads os.environ['HF_TOKEN'] but the Job never injected it -- the token could not actually reach the container. Add an env entry sourcing HF_TOKEN from the hf-token Secret with optional: true (Job still runs anonymously if the Secret is absent). Makes the recommendation functional. --- .../dreamzero/kubernetes/libero/README.md | 22 +++++++++++++------ .../kubernetes/libero/model-download.yaml | 11 ++++++++++ 2 files changed, 26 insertions(+), 7 deletions(-) diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md index b211f6383..4951fea46 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md @@ -148,8 +148,8 @@ Block for ML or an On-Demand Capacity Reservation (ODCR); the workload is fixed-size, so autoscaling is a convenience, not a requirement). You also need the NVIDIA GPU Operator (or device plugin) and EFA device plugin so pods can request `nvidia.com/gpu` and `vpc.amazonaws.com/efa`. Cluster-creation -references live in -[`1.architectures/4.amazon-eks`](../../../../../1.architectures/4.amazon-eks). +references live in the +[`Amazon EKS distributed training architecture`](../../../../../1.architectures/4.amazon-eks/README.md) section. Point your local kubeconfig at the cluster and confirm it is reachable: @@ -235,11 +235,17 @@ aws eks create-pod-identity-association \ --role-arn arn:aws:iam:::role/ ``` -> **Hugging Face auth is optional.** The default repos +> **Hugging Face auth is optional, but recommended.** The default repos > (`GEAR-Dreams/DreamZero-DROID`, `google/umt5-xxl`, -> `physical-intelligence/libero`) are **public** and download anonymously. Only -> if you must authenticate to a gated repo, create the `hf-token` Secret from -> `secret.example.yaml`: +> `physical-intelligence/libero`) are **public** and download anonymously, so a +> token is not *required*. However, anonymous downloads are rate-limited **per +> source IP** and the anonymous tier is substantially stricter than an +> authenticated one (HF's #1 cause of `429 Too Many Requests`). This download is +> large and multi-file (a 23B sharded checkpoint + tokenizer + video dataset), +> and on EKS it egresses through a shared NAT gateway — so the cluster's other +> workloads share that same per-IP anonymous quota. Providing an `HF_TOKEN` +> shifts you to a higher per-token quota and avoids that contention. Create the +> `hf-token` Secret (also required for gated repos): > > ```bash > kubectl -n "$NAMESPACE" create secret generic hf-token --from-literal=HF_TOKEN=hf_xxx @@ -328,7 +334,9 @@ This clones the pinned `RLinf` and `dreamzero` sources, builds stage 1 Downloads the DreamZero-DROID 14B warm-start checkpoint (`GEAR-Dreams/DreamZero-DROID`), the `google/umt5-xxl` tokenizer, and the -`physical-intelligence/libero` dataset (LeRobot layout) — all **anonymous**. This +`physical-intelligence/libero` dataset (LeRobot layout). The repos are public so +this works anonymously, but the Job picks up the optional `hf-token` Secret if +present (recommended — see the auth note above). This Job runs in a lightweight `python:3.11-slim` staging image, so plain `envsubst` is fine: diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/model-download.yaml b/3.test_cases/pytorch/dreamzero/kubernetes/libero/model-download.yaml index e98dc87a5..7c5b3b4be 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/model-download.yaml +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/model-download.yaml @@ -145,6 +145,17 @@ spec: echo "Dataset:" du -sh /fsx/datasets/libero 2>/dev/null || echo " (not found)" ls -la /fsx/models/DreamZero-DROID /fsx/models/umt5-xxl /fsx/datasets/libero + env: + # Optional Hugging Face token. The default repos are public (download + # works without it), but passing a token avoids the stricter + # anonymous per-IP rate limits and is required for gated repos. + # optional: true -> the Job runs fine if the hf-token Secret is absent. + - name: HF_TOKEN + valueFrom: + secretKeyRef: + name: hf-token + key: HF_TOKEN + optional: true resources: requests: cpu: "4" From f1ed257d32fca2faf00b771ecd68caa696359e00 Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Sun, 21 Jun 2026 14:35:55 -0500 Subject: [PATCH 42/45] docs(dreamzero): fix Step-by-step accuracy (RayJob logs + save_full cause) Final accuracy pass on the Step-by-step section: - Step 4: 'kubectl logs job/dreamzero-sft' was wrong -- dreamzero-sft is a RayJob, not a batch Job. KubeRay creates a submitter Job by that name whose logs only show 'ray job submit' plumbing; the training driver logs are on the Ray head pod. Use 'kubectl logs -l ray.io/node-type=head' instead. - Step 5: align the save_full_model_weights rationale with the accurate cause ('Backend nccl does not support allgather_into_tensor_coalesced') instead of the vaguer 'rank-0 gather stalls'. Verified accurate (no change needed): all job/ConfigMap names, the libero_sft_dreamzero_14b config name, eval LOG_DIR, convert STEP/output path, the video path, total_num_envs=16, and the dataset/metadata notes. --- .../pytorch/dreamzero/kubernetes/libero/README.md | 10 +++++++--- 1 file changed, 7 insertions(+), 3 deletions(-) diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md index 4951fea46..52e337c0a 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md @@ -384,7 +384,10 @@ kubectl -n "$NAMESPACE" create configmap dreamzero-sft-launcher \ envsubst '${ECR_URI} ${NAMESPACE}' < dreamzero-sft.yaml | kubectl apply -f - -kubectl logs -f -n "$NAMESPACE" job/dreamzero-sft +# dreamzero-sft is a RayJob. `kubectl logs job/...` would only show the KubeRay +# submitter's `ray job submit` output; the training driver logs live on the Ray +# head pod: +kubectl logs -f -n "$NAMESPACE" -l ray.io/node-type=head ``` The launcher drives `examples/sft/train_vla_sft.py` with config @@ -400,8 +403,9 @@ The launcher drives `examples/sft/train_vla_sft.py` with config ### 5. Convert the checkpoint (DCP shards → single `.pt`) Eval consumes a single consolidated `.pt`. Convert the sharded DCP offline on -**CPU** (do **not** use `save_full_model_weights` — the rank-0 full-state-dict -gather stalls on the 16B model). Create the launcher ConfigMap from +**CPU** (do **not** use `save_full_model_weights` — on the 16B model the +full-state-dict gather hits `Backend nccl does not support +allgather_into_tensor_coalesced`). Create the launcher ConfigMap from `scripts/convert_checkpoint.sh`, then apply with **restricted** `envsubst`: ```bash From 802bdf5ccbeba043f71a22ce8d9e82f7c71af5f2 Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Tue, 23 Jun 2026 09:43:33 -0500 Subject: [PATCH 43/45] docs(dreamzero): fix stale build-push.sh path in .gitignore comment The script moved out of setup/ to kubernetes/libero/build-push.sh (commit 58477a8); update the comment to match. Ignore rules unchanged. --- 3.test_cases/pytorch/dreamzero/.gitignore | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/3.test_cases/pytorch/dreamzero/.gitignore b/3.test_cases/pytorch/dreamzero/.gitignore index 31866ae0c..09f166c78 100644 --- a/3.test_cases/pytorch/dreamzero/.gitignore +++ b/3.test_cases/pytorch/dreamzero/.gitignore @@ -1,3 +1,3 @@ -# Transient build clones created by kubernetes/libero/setup/build-push.sh +# Transient build clones created by kubernetes/libero/build-push.sh /RLinf/ /DreamZero/ From 3262f6046c66ab96c002f49d2c326b5cdd658762 Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Tue, 23 Jun 2026 09:44:41 -0500 Subject: [PATCH 44/45] fix(dreamzero): use immutable IMAGE_TAG instead of mutable :latest A mutable :latest tag makes the pushed image non-reproducible and is the issue CONTRIBUTING calls out ("do not use a latest tag"). build-push.sh now defaults to an immutable IMAGE_TAG (dz-${DREAMZERO_REF}) and pushes ${ECR_URI}:${IMAGE_TAG}. The same IMAGE_TAG is threaded through every manifest's image reference and added to each restricted envsubst allow-list, plus env_vars.example and the README, so the pushed and deployed images are guaranteed to match. --- .../dreamzero/kubernetes/libero/README.md | 21 ++++++++++--------- .../dreamzero/kubernetes/libero/build-push.sh | 10 ++++++--- .../kubernetes/libero/convert-checkpoint.yaml | 5 +++-- .../kubernetes/libero/dreamzero-eval.yaml | 9 ++++---- .../kubernetes/libero/dreamzero-sft.yaml | 11 +++++----- .../kubernetes/libero/env_vars.example | 8 +++++++ .../kubernetes/libero/generate-metadata.yaml | 12 ++++++----- 7 files changed, 47 insertions(+), 29 deletions(-) diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md index 52e337c0a..f71c0545c 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/README.md @@ -300,13 +300,14 @@ The variables: | `AWS_REGION` | `us-east-1` | Region of your cluster / ECR. | | `UPSTREAM_REF` | `b3bbabb1f461` | Pinned RLinf commit baked into the image. | | `DREAMZERO_REF` | `ab790c198fbc` | Pinned `dreamzero` (`groot`) commit baked into the image. | +| `IMAGE_TAG` | `dz-ab790c198fbc` | Immutable tag for the training image. Used by both `build-push.sh` and the manifest `envsubst` so the pushed and deployed images always match (avoid a mutable `latest` tag). Defaults to `dz-${DREAMZERO_REF}`. | ## Step-by-step Throughout, `$NAMESPACE` and `$ECR_URI` come from `env_vars`. Wherever a manifest embeds inline shell, render it with the **restricted** allow-list -`envsubst '${ECR_URI} ${NAMESPACE}'` so the inline `${...}` shell variables are -not clobbered to empty strings. +`envsubst '${ECR_URI} ${NAMESPACE} ${IMAGE_TAG}'` so the inline `${...}` shell +variables are not clobbered to empty strings. ### 1. Build and push the training image @@ -318,8 +319,8 @@ external `groot` package is cloned from `github.com/RLinf/dreamzero.git` and placed on `PYTHONPATH` via `DREAMZERO_PATH=/workspace/DreamZero`. Build the image with **local Docker + buildx**. Run from the `kubernetes/libero/` -directory; it reads `ECR_URI`, `AWS_REGION`, `UPSTREAM_REF`, and `DREAMZERO_REF` -from `env_vars`: +directory; it reads `ECR_URI`, `AWS_REGION`, `UPSTREAM_REF`, `DREAMZERO_REF`, and +`IMAGE_TAG` from `env_vars`: ```bash source ./env_vars @@ -328,7 +329,7 @@ source ./env_vars This clones the pinned `RLinf` and `dreamzero` sources, builds stage 1 (`rlinf-upstream-embodied-libero`), then builds and pushes stage 2 to -`${ECR_URI}:latest`. +`${ECR_URI}:${IMAGE_TAG}`. ### 2. Stage models + dataset to FSx @@ -363,7 +364,7 @@ image** (it needs the RLinf toolkit + the `dreamzero` venv), so use **restricted `envsubst`: ```bash -envsubst '${ECR_URI} ${NAMESPACE}' < generate-metadata.yaml | kubectl apply -f - +envsubst '${ECR_URI} ${NAMESPACE} ${IMAGE_TAG}' < generate-metadata.yaml | kubectl apply -f - kubectl logs -f -n "$NAMESPACE" job/generate-metadata-dreamzero # -> writes /fsx/models/metadata-libero.json (top-level key `libero_sim`) ``` @@ -382,7 +383,7 @@ kubectl -n "$NAMESPACE" create configmap dreamzero-sft-launcher \ --from-file=run_dreamzero_sft_eks.sh=scripts/run_dreamzero_sft_eks.sh \ --dry-run=client -o yaml | kubectl apply -f - -envsubst '${ECR_URI} ${NAMESPACE}' < dreamzero-sft.yaml | kubectl apply -f - +envsubst '${ECR_URI} ${NAMESPACE} ${IMAGE_TAG}' < dreamzero-sft.yaml | kubectl apply -f - # dreamzero-sft is a RayJob. `kubectl logs job/...` would only show the KubeRay # submitter's `ray job submit` output; the training driver logs live on the Ray @@ -413,7 +414,7 @@ kubectl -n "$NAMESPACE" create configmap dreamzero-convert-launcher \ --from-file=convert_checkpoint.sh=scripts/convert_checkpoint.sh \ --dry-run=client -o yaml | kubectl apply -f - -envsubst '${ECR_URI} ${NAMESPACE}' < convert-checkpoint.yaml | kubectl apply -f - +envsubst '${ECR_URI} ${NAMESPACE} ${IMAGE_TAG}' < convert-checkpoint.yaml | kubectl apply -f - kubectl logs -f -n "$NAMESPACE" job/dreamzero-convert # -> .../global_step_1/actor/model_state_dict/full_weights.pt ``` @@ -437,7 +438,7 @@ kubectl -n "$NAMESPACE" create configmap dreamzero-eval-config \ --from-file=libero_spatial_eval_dreamzero_14b.yaml=scripts/libero_spatial_eval_dreamzero_14b.yaml \ --dry-run=client -o yaml | kubectl apply -f - -envsubst '${ECR_URI} ${NAMESPACE}' < dreamzero-eval.yaml | kubectl apply -f - +envsubst '${ECR_URI} ${NAMESPACE} ${IMAGE_TAG}' < dreamzero-eval.yaml | kubectl apply -f - kubectl logs -f -n "$NAMESPACE" job/dreamzero-eval ``` @@ -514,7 +515,7 @@ config: | Symptom | Cause | Fix | |---------|-------|-----| -| Pods crash with `ray: command not found`, or inline `${...}` vars render empty | Manifest rendered with unrestricted `envsubst`, clobbering inline shell vars | Always render manifests that embed shell with the **restricted** allow-list: `envsubst '${ECR_URI} ${NAMESPACE}' < ...`. | +| Pods crash with `ray: command not found`, or inline `${...}` vars render empty | Manifest rendered with unrestricted `envsubst`, clobbering inline shell vars | Always render manifests that embed shell with the **restricted** allow-list: `envsubst '${ECR_URI} ${NAMESPACE} ${IMAGE_TAG}' < ...`. | | `ray: command not found` on Ray head/worker/submitter | KubeRay runs `ray` non-interactively (`~/.bashrc` not sourced); `ray` lives in the `dreamzero` venv | Already handled in `dreamzero-sft.yaml`: the venv `bin` is prepended to `PATH` on the head/worker `env` **and** on the `submitterPodTemplate`. Don't strip those `PATH` entries. | | SFT crashes near checkpoint save: `UnpicklingError: invalid load key '\x00'` in `broadcast_object_list`, after shards are written | `dcp.save`'s finalization broadcast runs over the default NCCL/CUDA PG and races with NCCL teardown at the end of a long write (not torch-version-specific — the sync `dcp.save` path is unchanged through ≥ torch 2.8) | Fixed by the in-image `dcp-save-gloo-coordinator.patch`, which passes a dedicated gloo PG so the broadcast runs over CPU/gloo. If you somehow hit this, the **on-disk checkpoint is still valid and convertible** — proceed to step 5 (convert). | | Convert fails with `EOFError` / `inline_container.cc unexpected pos`, corrupt shards | FSx filled up during the SFT run; `torch.save` was truncated mid-write | Ensure **≥250 GB free** on FSx before SFT (a 14B DCP checkpoint is ~140–206 GB). Free space, re-run SFT. | diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/build-push.sh b/3.test_cases/pytorch/dreamzero/kubernetes/libero/build-push.sh index 7b17e3d59..77a30421a 100755 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/build-push.sh +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/build-push.sh @@ -17,7 +17,11 @@ UPSTREAM_REF="${UPSTREAM_REF:-b3bbabb1f461}" DREAMZERO_REPO="${DREAMZERO_REPO:-https://github.com/RLinf/dreamzero.git}" DREAMZERO_REF="${DREAMZERO_REF:-ab790c198fbc}" BUILD_TARGET="${BUILD_TARGET:-embodied-libero}" -TAG="${TAG:-latest}" +# Immutable image tag. Defaults to one derived from the pinned DreamZero ref so the +# pushed image is reproducible (CONTRIBUTING: "do not use a latest tag"). The SAME +# value must be exported as IMAGE_TAG when rendering the manifests so the deployed +# image matches the pushed one -- see env_vars.example. +IMAGE_TAG="${IMAGE_TAG:-dz-${DREAMZERO_REF}}" # Test-case root (this script is at kubernetes/libero/build-push.sh). ROOT="$(cd "$(dirname "$0")/../.." && pwd)" @@ -41,9 +45,9 @@ docker buildx build --platform linux/amd64 --load \ echo "== Stage 2: EFA overlay (push) ==" docker buildx build --platform linux/amd64 --push \ --build-arg BUILD_TARGET="$BUILD_TARGET" \ - -t "${ECR_URI}:${TAG}" \ + -t "${ECR_URI}:${IMAGE_TAG}" \ -f Dockerfile . -echo "== Done: ${ECR_URI}:${TAG} ==" +echo "== Done: ${ECR_URI}:${IMAGE_TAG} ==" echo "== Cleaning transient clones ==" rm -rf RLinf DreamZero diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/convert-checkpoint.yaml b/3.test_cases/pytorch/dreamzero/kubernetes/libero/convert-checkpoint.yaml index 36b008715..00b73a508 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/convert-checkpoint.yaml +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/convert-checkpoint.yaml @@ -17,11 +17,12 @@ # Usage (restricted envsubst protects the inline ${...} shell vars): # export ECR_URI=.dkr.ecr..amazonaws.com/ # export NAMESPACE=dreamzero +# export IMAGE_TAG=dz-ab790c198fbc # immutable tag pushed by build-push.sh # # Optionally override STEP (default global_step_1): # kubectl -n $NAMESPACE create configmap dreamzero-convert-launcher \ # --from-file=convert_checkpoint.sh=kubernetes/libero/scripts/convert_checkpoint.sh \ # --dry-run=client -o yaml | kubectl apply -f - -# envsubst '${ECR_URI} ${NAMESPACE}' < kubernetes/libero/convert-checkpoint.yaml | kubectl apply -f - +# envsubst '${ECR_URI} ${NAMESPACE} ${IMAGE_TAG}' < kubernetes/libero/convert-checkpoint.yaml | kubectl apply -f - --- apiVersion: batch/v1 kind: Job @@ -42,7 +43,7 @@ spec: restartPolicy: Never containers: - name: convert - image: ${ECR_URI}:latest + image: ${ECR_URI}:${IMAGE_TAG} command: ["bash", "/opt/scripts/convert_checkpoint.sh"] env: - name: VENV_NAME diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-eval.yaml b/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-eval.yaml index 6ba80f1be..7a49ad750 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-eval.yaml +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-eval.yaml @@ -33,11 +33,12 @@ # --from-file=libero_spatial_eval_dreamzero_14b.yaml=kubernetes/libero/scripts/libero_spatial_eval_dreamzero_14b.yaml \ # --dry-run=client -o yaml | kubectl apply -f - # -# Apply (RESTRICTED envsubst -- substitute ONLY ${ECR_URI} and ${NAMESPACE} so the -# inline ${PYTHONPATH:-} guards in the bootstrap are NOT clobbered): +# Apply (RESTRICTED envsubst -- substitute ONLY ${ECR_URI}, ${NAMESPACE} and +# ${IMAGE_TAG} so the inline ${PYTHONPATH:-} guards in the bootstrap are NOT clobbered): # export ECR_URI=.dkr.ecr..amazonaws.com/ # export NAMESPACE=dreamzero -# envsubst '${ECR_URI} ${NAMESPACE}' < kubernetes/libero/dreamzero-eval.yaml | kubectl apply -f - +# export IMAGE_TAG=dz-ab790c198fbc # immutable tag pushed by build-push.sh +# envsubst '${ECR_URI} ${NAMESPACE} ${IMAGE_TAG}' < kubernetes/libero/dreamzero-eval.yaml | kubectl apply -f - --- apiVersion: batch/v1 kind: Job @@ -69,7 +70,7 @@ spec: terminationGracePeriodSeconds: 120 containers: - name: eval - image: ${ECR_URI}:latest + image: ${ECR_URI}:${IMAGE_TAG} command: - bash - /opt/scripts/run_dreamzero_eval_eks.sh diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-sft.yaml b/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-sft.yaml index 03ad2a189..9165e5ec1 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-sft.yaml +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-sft.yaml @@ -29,10 +29,11 @@ # --dry-run=client -o yaml | kubectl apply -f - # export ECR_URI=.dkr.ecr..amazonaws.com/ # export NAMESPACE=dreamzero -# # RESTRICTED envsubst -- substitute ONLY ${ECR_URI} and ${NAMESPACE}. +# export IMAGE_TAG=dz-ab790c198fbc # immutable tag pushed by build-push.sh +# # RESTRICTED envsubst -- substitute ONLY ${ECR_URI}, ${NAMESPACE} and ${IMAGE_TAG}. # # (The Ray head-election bash is gone, but keep this habit: other inline # # shell snippets, e.g. the launcher copy, must not be expanded.) -# envsubst '${ECR_URI} ${NAMESPACE}' < kubernetes/libero/dreamzero-sft.yaml | kubectl apply -f - +# envsubst '${ECR_URI} ${NAMESPACE} ${IMAGE_TAG}' < kubernetes/libero/dreamzero-sft.yaml | kubectl apply -f - --- apiVersion: ray.io/v1 kind: RayJob @@ -66,7 +67,7 @@ spec: restartPolicy: Never containers: - name: ray-job-submitter - image: ${ECR_URI}:latest + image: ${ECR_URI}:${IMAGE_TAG} env: - name: PATH value: "/opt/venv/dreamzero/bin:/opt/amazon/openmpi/bin:/opt/amazon/efa/bin:/opt/gdrcopy/bin:/usr/local/nvidia/bin:/usr/local/cuda/bin:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin" @@ -102,7 +103,7 @@ spec: topologyKey: kubernetes.io/hostname containers: - name: ray-head - image: ${ECR_URI}:latest + image: ${ECR_URI}:${IMAGE_TAG} env: &dreamzero_env # KubeRay runs `ray start` as the container command (non-interactive, # ~/.bashrc not sourced). `ray` lives in the dreamzero venv, so prepend @@ -207,7 +208,7 @@ spec: topologyKey: kubernetes.io/hostname containers: - name: ray-worker - image: ${ECR_URI}:latest + image: ${ECR_URI}:${IMAGE_TAG} env: *dreamzero_env resources: *dreamzero_resources volumeMounts: *dreamzero_mounts diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/env_vars.example b/3.test_cases/pytorch/dreamzero/kubernetes/libero/env_vars.example index d536d0e86..f658796f8 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/env_vars.example +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/env_vars.example @@ -17,6 +17,14 @@ export AWS_REGION=us-east-1 export UPSTREAM_REF=b3bbabb1f461 export DREAMZERO_REF=ab790c198fbc +# Immutable image tag for the training image. This SAME value is used both when +# building/pushing (build-push.sh) and when rendering the manifests (the envsubst +# commands below substitute ${IMAGE_TAG}), so the pushed and deployed images always +# match. Avoid a mutable `latest` tag (CONTRIBUTING: "do not use a latest tag"). +# build-push.sh defaults this to dz-${DREAMZERO_REF} if unset; set it explicitly +# here to keep build and deploy in lockstep. +export IMAGE_TAG=dz-ab790c198fbc + # --------------------------------------------------------------------------- # OPTIONAL: only needed if you provision FSx via the storage/ manifests. # Skip these if you already have an `fsx-claim` PVC bound in your namespace. diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/generate-metadata.yaml b/3.test_cases/pytorch/dreamzero/kubernetes/libero/generate-metadata.yaml index e17c90785..2f2d90392 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/generate-metadata.yaml +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/generate-metadata.yaml @@ -18,7 +18,7 @@ # Metadata generation requires the RLinf repo's toolkit # (toolkits/lerobot/generate_dreamzero_metadata.py) PLUS the dreamzero venv # (groot, lerobot, etc.). The slim staging image used by model-download.yaml -# does NOT have these, so this step runs in ${ECR_URI}:latest instead. +# does NOT have these, so this step runs in ${ECR_URI}:${IMAGE_TAG} instead. # # OUTPUT: # Writes /fsx/models/metadata-libero.json (top-level key `libero_sim`). The SFT @@ -27,17 +27,19 @@ # # Prerequisites: # - LIBERO dataset staged on FSx at /fsx/datasets/libero (by model-download.yaml) -# - Training image built and pushed to ECR (${ECR_URI}:latest) +# - Training image built and pushed to ECR (${ECR_URI}:${IMAGE_TAG}) # - FSx PVC "fsx-claim" bound in the target namespace # # Usage: # export ECR_URI=.dkr.ecr..amazonaws.com/ # export NAMESPACE=dreamzero -# # NOTE: use RESTRICTED envsubst -- substitute ONLY ${ECR_URI} and ${NAMESPACE}. +# export IMAGE_TAG=dz-ab790c198fbc # immutable tag pushed by build-push.sh +# # NOTE: use RESTRICTED envsubst -- substitute ONLY ${ECR_URI}, ${NAMESPACE} and +# # ${IMAGE_TAG}. # # The container command contains shell vars (${PYTHONPATH}, ${DREAMZERO_PATH}) # # that unrestricted envsubst would clobber to empty strings. Same pattern as # # dreamzero-sft.yaml. -# envsubst '${ECR_URI} ${NAMESPACE}' < kubernetes/libero/generate-metadata.yaml | kubectl apply -f - +# envsubst '${ECR_URI} ${NAMESPACE} ${IMAGE_TAG}' < kubernetes/libero/generate-metadata.yaml | kubectl apply -f - # kubectl logs -f job/generate-metadata-dreamzero -n $NAMESPACE --- apiVersion: batch/v1 @@ -61,7 +63,7 @@ spec: restartPolicy: Never containers: - name: generate-metadata - image: ${ECR_URI}:latest + image: ${ECR_URI}:${IMAGE_TAG} command: - bash - -lc From b14e0ad375dd1df78a2747ff20529c34d9e0bf85 Mon Sep 17 00:00:00 2001 From: bluecrayon52 <16687465+bluecrayon52@users.noreply.github.com> Date: Tue, 23 Jun 2026 09:44:48 -0500 Subject: [PATCH 45/45] refactor(dreamzero): standardize launcher mount on /opt/scripts SFT mounted its launcher ConfigMap at /tmp/scripts while convert and eval use /opt/scripts. Standardize SFT on /opt/scripts (and the matching entrypoint paths) so the recipe is consistent and easier to copy-adapt between steps. --- .../dreamzero/kubernetes/libero/dreamzero-sft.yaml | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-sft.yaml b/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-sft.yaml index 9165e5ec1..164f953ce 100644 --- a/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-sft.yaml +++ b/3.test_cases/pytorch/dreamzero/kubernetes/libero/dreamzero-sft.yaml @@ -46,14 +46,14 @@ metadata: example: dreamzero-sft spec: # Run the launcher on the head once the Ray cluster is ready. The launcher is - # mounted read-only from the dreamzero-sft-launcher ConfigMap at /tmp/scripts; + # mounted read-only from the dreamzero-sft-launcher ConfigMap at /opt/scripts; # run it directly from there (it is path-independent -- it cd's to # /workspace/RLinf itself). A fallback to a baked-in copy is kept for images # that ship the launcher under /workspace/eks/scripts. entrypoint: >- bash -c ' - if [ -f /tmp/scripts/run_dreamzero_sft_eks.sh ]; then - bash /tmp/scripts/run_dreamzero_sft_eks.sh; + if [ -f /opt/scripts/run_dreamzero_sft_eks.sh ]; then + bash /opt/scripts/run_dreamzero_sft_eks.sh; else bash /workspace/eks/scripts/run_dreamzero_sft_eks.sh; fi' @@ -161,7 +161,7 @@ spec: - name: dshm mountPath: /dev/shm - name: launcher - mountPath: /tmp/scripts + mountPath: /opt/scripts readOnly: true volumes: &dreamzero_volumes - name: fsx