SLURM_JOB_ID = 1038301 SLURM_JOB_NAME = nvr_elm_llm:train/NVILA-Lite-8B-quantumn-qa-train RUN_NAME = NVILA-Lite-8B-quantumn-qa-train OUTPUT_DIR = runs/train/NVILA-Lite-8B-quantumn-qa-train NNODES = 8 NODES = pool0-02107 pool0-02117 pool0-02099 pool0-02196 pool0-02404 pool0-02496 pool0-02566 pool0-02669 NODE_RANK = 1 GPUS_PER_NODE = 8 MASTER_ADDR = pool0-02107 MASTER_PORT = 25001 GLOBAL_TRAIN_BATCH_SIZE = 2048 GRADIENT_ACCUMULATION_STEPS = 4 PER_DEVICE_TRAIN_BATCH_SIZE = 8 DEFAULT_LEARNING_RATE: 2e-5 SLURM_JOB_ID = 1038301 SLURM_JOB_NAME = nvr_elm_llm:train/NVILA-Lite-8B-quantumn-qa-train RUN_NAME = NVILA-Lite-8B-quantumn-qa-train OUTPUT_DIR = runs/train/NVILA-Lite-8B-quantumn-qa-train NNODES = 8 NODES = pool0-02107 pool0-02117 pool0-02099 pool0-02196 pool0-02404 pool0-02496 pool0-02566 pool0-02669 NODE_RANK = 5 GPUS_PER_NODE = 8 MASTER_ADDR = pool0-02107 MASTER_PORT = 25001 GLOBAL_TRAIN_BATCH_SIZE = 2048 GRADIENT_ACCUMULATION_STEPS = 4 PER_DEVICE_TRAIN_BATCH_SIZE = 8 DEFAULT_LEARNING_RATE: 2e-5 SLURM_JOB_ID = 1038301 SLURM_JOB_NAME = nvr_elm_llm:train/NVILA-Lite-8B-quantumn-qa-train RUN_NAME = NVILA-Lite-8B-quantumn-qa-train OUTPUT_DIR = runs/train/NVILA-Lite-8B-quantumn-qa-train NNODES = 8 NODES = pool0-02107 pool0-02117 pool0-02099 pool0-02196 pool0-02404 pool0-02496 pool0-02566 pool0-02669 NODE_RANK = 0 GPUS_PER_NODE = 8 MASTER_ADDR = pool0-02107 MASTER_PORT = 25001 GLOBAL_TRAIN_BATCH_SIZE = 2048 GRADIENT_ACCUMULATION_STEPS = 4 PER_DEVICE_TRAIN_BATCH_SIZE = 8 DEFAULT_LEARNING_RATE: 2e-5 SLURM_JOB_ID = 1038301 SLURM_JOB_NAME = nvr_elm_llm:train/NVILA-Lite-8B-quantumn-qa-train RUN_NAME = NVILA-Lite-8B-quantumn-qa-train OUTPUT_DIR = runs/train/NVILA-Lite-8B-quantumn-qa-train NNODES = 8 NODES = pool0-02107 pool0-02117 pool0-02099 pool0-02196 pool0-02404 pool0-02496 pool0-02566 pool0-02669 NODE_RANK = 2 GPUS_PER_NODE = 8 MASTER_ADDR = pool0-02107 MASTER_PORT = 25001 GLOBAL_TRAIN_BATCH_SIZE = 2048 GRADIENT_ACCUMULATION_STEPS = 4 PER_DEVICE_TRAIN_BATCH_SIZE = 8 DEFAULT_LEARNING_RATE: 2e-5 SLURM_JOB_ID = 1038301 SLURM_JOB_NAME = nvr_elm_llm:train/NVILA-Lite-8B-quantumn-qa-train RUN_NAME = NVILA-Lite-8B-quantumn-qa-train OUTPUT_DIR = runs/train/NVILA-Lite-8B-quantumn-qa-train NNODES = 8 NODES = pool0-02107 pool0-02117 pool0-02099 pool0-02196 pool0-02404 pool0-02496 pool0-02566 pool0-02669 NODE_RANK = 7 GPUS_PER_NODE = 8 MASTER_ADDR = pool0-02107 MASTER_PORT = 25001 GLOBAL_TRAIN_BATCH_SIZE = 2048 GRADIENT_ACCUMULATION_STEPS = 4 PER_DEVICE_TRAIN_BATCH_SIZE = 8 DEFAULT_LEARNING_RATE: 2e-5 SLURM_JOB_ID = 1038301 SLURM_JOB_NAME = nvr_elm_llm:train/NVILA-Lite-8B-quantumn-qa-train RUN_NAME = NVILA-Lite-8B-quantumn-qa-train OUTPUT_DIR = runs/train/NVILA-Lite-8B-quantumn-qa-train NNODES = 8 NODES = pool0-02107 pool0-02117 pool0-02099 pool0-02196 pool0-02404 pool0-02496 pool0-02566 pool0-02669 NODE_RANK = 6 GPUS_PER_NODE = 8 MASTER_ADDR = pool0-02107 MASTER_PORT = 25001 GLOBAL_TRAIN_BATCH_SIZE = 2048 GRADIENT_ACCUMULATION_STEPS = 4 PER_DEVICE_TRAIN_BATCH_SIZE = 8 DEFAULT_LEARNING_RATE: 2e-5 SLURM_JOB_ID = 1038301 SLURM_JOB_NAME = nvr_elm_llm:train/NVILA-Lite-8B-quantumn-qa-train RUN_NAME = NVILA-Lite-8B-quantumn-qa-train OUTPUT_DIR = runs/train/NVILA-Lite-8B-quantumn-qa-train NNODES = 8 NODES = pool0-02107 pool0-02117 pool0-02099 pool0-02196 pool0-02404 pool0-02496 pool0-02566 pool0-02669 NODE_RANK = 3 GPUS_PER_NODE = 8 MASTER_ADDR = pool0-02107 MASTER_PORT = 25001 GLOBAL_TRAIN_BATCH_SIZE = 2048 GRADIENT_ACCUMULATION_STEPS = 4 PER_DEVICE_TRAIN_BATCH_SIZE = 8 DEFAULT_LEARNING_RATE: 2e-5 SLURM_JOB_ID = 1038301 SLURM_JOB_NAME = nvr_elm_llm:train/NVILA-Lite-8B-quantumn-qa-train RUN_NAME = NVILA-Lite-8B-quantumn-qa-train OUTPUT_DIR = runs/train/NVILA-Lite-8B-quantumn-qa-train NNODES = 8 NODES = pool0-02107 pool0-02117 pool0-02099 pool0-02196 pool0-02404 pool0-02496 pool0-02566 pool0-02669 NODE_RANK = 4 GPUS_PER_NODE = 8 MASTER_ADDR = pool0-02107 MASTER_PORT = 25001 GLOBAL_TRAIN_BATCH_SIZE = 2048 GRADIENT_ACCUMULATION_STEPS = 4 PER_DEVICE_TRAIN_BATCH_SIZE = 8 DEFAULT_LEARNING_RATE: 2e-5