Remove internal lustre paths (#1148)

Remove internal /lusre paths

<!-- This is an auto-generated comment: release notes by coderabbit.ai
-->
## Summary by CodeRabbit

* **Configuration Updates**
* Set a concrete default container image to an NVIDIA PyTorch container
for templates.
* Removed built-in default values in scripts; key environment/config
variables must be provided externally.
* Replaced absolute example paths with placeholder relative paths and
TODO notes requiring user customization.
<!-- end of auto-generated comment: release notes by coderabbit.ai -->

---------

Signed-off-by: Keval Morabia <28916987+kevalmorabia97@users.noreply.github.com>
This commit is contained in:
Keval Morabia
2026-04-01 17:53:20 +05:30
committed by GitHub
parent c37c74f651
commit d6c8e9d2b6
4 changed files with 5 additions and 16 deletions
@@ -61,7 +61,7 @@ export DATACACHE_DIR="" # path to data cache directory
########################################################
# CONTAINER
########################################################
export CONTAINER_IMAGE="" # path to container image, e.g., nvcr.io/nvidia/pytorch:26.01-py3
export CONTAINER_IMAGE="nvcr.io/nvidia/pytorch:26.01-py3" # path to container image or .sqsh file
export CONTAINER_MOUNTS="" # container mounts, e.g., "/lustre/fs1:/lustre/fs1"
export CONTAINER_WORKDIR="" # container work directory, e.g., "<path-to-modelopt>/Model-Optimizer/examples/llm_qad"
@@ -59,7 +59,7 @@ export DATACACHE_DIR="" # path to data cache directory
########################################################
# CONTAINER
########################################################
export CONTAINER_IMAGE="" # path to container image, e.g., nvcr.io/nvidia/pytorch:26.01-py3
export CONTAINER_IMAGE="nvcr.io/nvidia/pytorch:26.01-py3" # path to container image or .sqsh file
export CONTAINER_MOUNTS="" # container mounts, e.g., "/lustre/fs1:/lustre/fs1"
export CONTAINER_WORKDIR="" # container work directory
-11
View File
@@ -58,19 +58,8 @@ if [[ -n "$CONFIG_FILE" ]]; then
fi
fi
# === Default Paths (override in config) ===
MLM_DIR="${MLM_DIR:-/lustre/fs1/portfolios/coreai/projects/coreai_dlalgo_modelopt/users/weimingc/workspace/Megatron-LM}"
MODELOPT_DIR="${MODELOPT_DIR:-/lustre/fs1/portfolios/coreai/projects/coreai_dlalgo_modelopt/users/weimingc/workspace/TensorRT-Model-Optimizer}"
MODELS_ROOT="${MODELS_ROOT:-/lustre/fs1/portfolios/coreai/projects/coreai_dlalgo_modelopt/users/weimingc/models}"
QAD_CHECKPOINT_ROOT="${QAD_CHECKPOINT_ROOT:-/lustre/fs1/portfolios/coreai/projects/coreai_dlalgo_modelopt/users/weimingc/checkpoints}"
DATACACHE_DIR="${DATACACHE_DIR:-/lustre/fs1/portfolios/coreai/projects/coreai_dlalgo_modelopt/users/weimingc/data_cache}"
LOG_DIR="${LOG_DIR:-${QAD_CHECKPOINT_ROOT}/logs_slurm}"
# Container settings
CONTAINER_IMAGE="${CONTAINER_IMAGE:-/lustre/fs1/portfolios/coreai/projects/coreai_dlalgo_modelopt/users/weimingc/containers/pytorch_25.06-py3.sqsh}"
CONTAINER_MOUNTS="${CONTAINER_MOUNTS:-/lustre/fs1:/lustre/fs1}"
CONTAINER_WORKDIR="${CONTAINER_WORKDIR:-/lustre/fs1/portfolios/coreai/projects/coreai_dlalgo_modelopt/users/weimingc/workspace/TensorRT-Model-Optimizer/examples/llm_qad}"
# Parallelism (required from config)
TP_SIZE="${TP_SIZE:?ERROR: TP_SIZE must be set in config}"
MBS="${MBS:?ERROR: MBS must be set in config}"
@@ -1,9 +1,9 @@
# LTX-2 QAD Training Configuration
model:
model_path: "/lustre/fsw/portfolios/adlr/projects/adlr_psx_numerics/users/ynankani/ComfyUI/models/checkpoints/ltx-av-step-1933500-split-new-vae.safetensors"
model_path: "/path/to/ltx2/checkpoint.safetensors" # TODO: Set your LTX-2 checkpoint path
training_mode: "full"
load_checkpoint:
text_encoder_path: "/lustre/fsw/portfolios/adlr/users/dhutchins/models/gemma"
text_encoder_path: "/path/to/gemma" # TODO: Set your Gemma text encoder path
training_strategy:
name: "text_to_video"
@@ -26,7 +26,7 @@ acceleration:
load_text_encoder_in_8bit: true
data:
preprocessed_data_root: "/lustre/fsw/portfolios/adlr/users/scavallari/ltx-qad/qad-dataset"
preprocessed_data_root: "/path/to/preprocessed" # TODO: Set your preprocessed dataset path
num_dataloader_workers: 2
validation: