Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
20 changes: 20 additions & 0 deletions training/DeepSpeed-Reflow/.gitignore
Original file line number Diff line number Diff line change
@@ -0,0 +1,20 @@
# Conda environment setup — local to our environment, not part of the public example.
# Public users install via requirements.txt; this captures our exact dev environment.
environment.yml
conda_env*.yml
*.conda.yml

# Generated at runtime by the launcher scripts (DeepSpeed config + training outputs/logs).
*_config.json
*_output/
__pycache__/
wandb/

# JIT-built CPU-Adam/Lion op extensions (torch cpp_extension build dirs).
cpu_adam/
cpu_lion/
fused_adam/

# Comparison / benchmark output collected by run_compare.sh.
compare_out/
bitexact_out/
214 changes: 214 additions & 0 deletions training/DeepSpeed-Reflow/README.md

Large diffs are not rendered by default.

117 changes: 117 additions & 0 deletions training/DeepSpeed-Reflow/check_bitexact.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,117 @@
#!/bin/bash
# SPDX-License-Identifier: Apache-2.0
# DeepSpeed Team
# check_bitexact.sh -- verify that Reflow is bit-identical to ZeRO-Infinity
# (optimizer offload + ZeRO Stage 3).
#
# Usage: bash check_bitexact.sh [num_gpus] (default: 1)
#
# Runs the same OPT-350M job twice, once with each mode, under full
# determinism (--loss_check), then diffs the per-step hex-float losses.
# Everything is self-contained: the two configs are written here and
# differ only in the Reflow keys.
#
# --loss_check pins the NCCL collectives and forces deterministic
# algorithms, so this is slower than a normal run. It is for
# verification only -- speed and memory are measured by run_compare.sh.

set -euo pipefail
SCRIPT_DIR=$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd)

NUM_GPUS=${1:-1}
BATCH_SIZE=${BATCH_SIZE:-4}
OPTIMIZER=${OPTIMIZER:-adam}
BENCH_STEPS=${BENCH_STEPS:-6}
MODEL_NAME=${MODEL_NAME:-facebook/opt-350m}

# CPU affinity knobs, as in the finetune_*.sh scripts. Keep
# main_thread_cores + bucketwise_cores_per_worker within the cores
# visible to each rank.
MAIN_CORES=${MAIN_CORES:-2}
WORKER_CORES=${WORKER_CORES:-8}

OUT_DIR=${OUT_DIR:-bitexact_out}
for value in "$NUM_GPUS" "$BATCH_SIZE" "$BENCH_STEPS" "$MAIN_CORES" "$WORKER_CORES"; do
if [[ ! "$value" =~ ^[1-9][0-9]*$ ]]; then
echo "GPU, batch, step and core counts must be positive integers" >&2
exit 2
fi
done
mkdir -p "$OUT_DIR"

# ---------------------------------------------------------------- configs
# Same template as the finetune_*.sh scripts. overlap_comm stays false:
# with overlapped reductions the reduction order becomes timing-dependent
# and the two modes can legitimately diverge in the last bits.
write_config() { # $1 = path, $2 = reflow|zerooffload
if [ "$2" = "reflow" ]; then
REFLOW_BLOCK=',
"reflow": {
"enable_cpu_affinity": true,
"bucketwise_cores_per_worker": '"$WORKER_CORES"',
"main_thread_cores": '"$MAIN_CORES"'
}'
else
REFLOW_BLOCK=""
fi
cat > "$1" << EOF
{
"train_micro_batch_size_per_gpu": $BATCH_SIZE,
"gradient_accumulation_steps": 1,
"gradient_clipping": 0.0,
"bf16": { "enabled": true },
"zero_optimization": {
"stage": 3,
"overlap_comm": false,
"reduce_bucket_size": 4e8,
"sub_group_size": 4e8,
"offload_optimizer": {
"device": "cpu",
"pin_memory": true
}$REFLOW_BLOCK
},
"wall_clock_breakdown": true
}
EOF
}

REFLOW_CFG="$OUT_DIR/opt-350m_reflow_config.json"
ZERO_CFG="$OUT_DIR/opt-350m_zerooffload_config.json"
write_config "$REFLOW_CFG" reflow
write_config "$ZERO_CFG" zerooffload

# ------------------------------------------------------------------- runs
# Eager attention is required: flash-attention's backward is
# non-deterministic and would mask a real difference.
COMMON=(--model_name "$MODEL_NAME"
--optimizer "$OPTIMIZER" --loss_check --attn_implementation eager
--lr 1e-5 --batch_size "$BATCH_SIZE" --max_length 512
--bench_steps "$BENCH_STEPS" --warmup_steps 0 --output_dir "$OUT_DIR/out")

echo "[1/2] Reflow ..."
deepspeed --num_gpus="$NUM_GPUS" --bind_cores_to_rank "$SCRIPT_DIR/finetune_zero3.py" \
--deepspeed_config="$REFLOW_CFG" \
"${COMMON[@]}" > "$OUT_DIR/reflow.log" 2>&1

echo "[2/2] ZeRO-Infinity baseline ..."
deepspeed --num_gpus="$NUM_GPUS" --bind_cores_to_rank "$SCRIPT_DIR/finetune_zero3.py" \
--deepspeed_config="$ZERO_CFG" \
"${COMMON[@]}" > "$OUT_DIR/zero.log" 2>&1

# Ignore timestamps but require both jobs to have completed every requested step.
python3 - "$OUT_DIR/reflow.log" "$OUT_DIR/zero.log" "$BENCH_STEPS" <<'PYTHON'
import re
import sys
from pathlib import Path

expected_steps = list(range(1, int(sys.argv[3]) + 1))
losses = []
for filename in sys.argv[1:3]:
matches = re.findall(r"BITLOSS step (\d+) hex=(\S+)", Path(filename).read_text())
if [int(step) for step, _ in matches] != expected_steps:
sys.exit(f"Incomplete loss log: {filename}; expected steps 1 through {expected_steps[-1]}")
losses.append([value for _, value in matches])
if losses[0] != losses[1]:
sys.exit(f"MISMATCH -- losses differ; see {sys.argv[1]} and {sys.argv[2]}")
print("BIT-IDENTICAL")
PYTHON
153 changes: 153 additions & 0 deletions training/DeepSpeed-Reflow/finetune_llama-13b_1gpu.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,153 @@
#!/bin/bash
# SPDX-License-Identifier: Apache-2.0
# DeepSpeed Team
set -euo pipefail

echo "================================================"
echo "Llama 13B Fine-tuning with DeepSpeed Reflow on 1 GPU"
echo "================================================"

# MODE options: "reflow" or "zerooffload"
MODE=${1:-}
BATCH_SIZE=${2:-4}

GPUS_PER_NODE=1
if [[ "$MODE" != "reflow" && "$MODE" != "zerooffload" ]]; then
echo "Usage: bash $0 <reflow|zerooffload> [global_batch_size]" >&2
exit 2
fi
if [[ ! "$BATCH_SIZE" =~ ^[1-9][0-9]*$ ]] || (( BATCH_SIZE % GPUS_PER_NODE != 0 )); then
echo "global_batch_size must be positive and divisible by $GPUS_PER_NODE" >&2
exit 2
fi
MICRO_BATCH_SIZE=$((BATCH_SIZE / GPUS_PER_NODE))

SCRIPT_DIR=$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd)
MODEL_NAME="meta-llama/Llama-2-13b-hf"
OUTPUT_DIR="${SCRIPT_DIR}/llama-13b_${MODE}_output"
DS_CONFIG_JSON="${SCRIPT_DIR}/llama-13b_${MODE}_config.json"

mkdir -p "$OUTPUT_DIR"

# Script argument parameters
ACTIVATION_CHECKPOINTING=true
SAVE_CHECKPOINT=false
MAX_LENGTH=${MAX_LENGTH:-2048}
ATTN_IMPLEMENTATION=${ATTN_IMPLEMENTATION:-flash_attention_2}
LOG_INTERVAL=1
DATASET_NAME="tatsu-lab/alpaca"
DATASET_PERCENTAGE=10.0
USE_WANDB=false
WANDB_PROJECT="llama-13b"
WANDB_RUN_NAME="llama-13b-$MODE"
DETERMINISTIC=false
BENCH_STEPS=${BENCH_STEPS:-10}
WARMUP_STEPS=${WARMUP_STEPS:-1}

EPOCHS=1
LR=1e-5
WEIGHT_DECAY=0.01
SEED=42

ACTIVATION_CHECKPOINTING_FLAG=""
if [ "$ACTIVATION_CHECKPOINTING" = "true" ]; then
ACTIVATION_CHECKPOINTING_FLAG="--activation_checkpointing"
fi

SAVE_CHECKPOINT_ARG=""
if [ "$SAVE_CHECKPOINT" = "true" ]; then
SAVE_CHECKPOINT_ARG="--save_checkpoint"
fi

WANDB_FLAG=""
if [ "$USE_WANDB" = "true" ]; then
WANDB_FLAG="--use_wandb"
fi

DETERMINISTIC_FLAG=""
if [ "$DETERMINISTIC" = "true" ]; then
DETERMINISTIC_FLAG="--deterministic"
fi

# Reflow uses the same ZeRO-3 configuration as the baseline plus the reflow block.
if [ "$MODE" = "reflow" ]; then
cat > "$DS_CONFIG_JSON" << EOF
{
"train_batch_size": $BATCH_SIZE,
"gradient_accumulation_steps": 1,
"gradient_clipping": 0.0,
"bf16": { "enabled": true },
"zero_optimization": {
"stage": 3,
"overlap_comm": false,
"reduce_bucket_size": 4e8,
"sub_group_size": 1e9,
"offload_optimizer": {
"device": "cpu",
"pin_memory": true
},
"reflow": {
"enable_cpu_affinity": true,
"bucketwise_cores_per_worker": ${WORKER_CORES:-8},
"main_thread_cores": ${MAIN_CORES:-2}
}
},
"wall_clock_breakdown": true
}
EOF

elif [ "$MODE" = "zerooffload" ]; then
cat > "$DS_CONFIG_JSON" << EOF
{
"train_batch_size": $BATCH_SIZE,
"gradient_accumulation_steps": 1,
"gradient_clipping": 0.0,
"bf16": { "enabled": true },
"zero_optimization": {
"stage": 3,
"overlap_comm": false,
"reduce_bucket_size": 4e8,
"sub_group_size": 1e9,
"offload_optimizer": {
"device": "cpu",
"pin_memory": true
}
},
"wall_clock_breakdown": true
}
EOF
fi


CMD=(
deepspeed --num_gpus=$GPUS_PER_NODE --bind_cores_to_rank "$SCRIPT_DIR/finetune_zero3.py"
--deepspeed_config="$DS_CONFIG_JSON"
--model_name "$MODEL_NAME"
--attn_implementation "$ATTN_IMPLEMENTATION"
--num_train_epochs "$EPOCHS"
--lr "$LR"
--batch_size "$MICRO_BATCH_SIZE"
--weight_decay "$WEIGHT_DECAY"
--output_dir "$OUTPUT_DIR"
--seed "$SEED"
--max_length "$MAX_LENGTH"
--log_interval "$LOG_INTERVAL"
--dataset_name "$DATASET_NAME"
--dataset_percentage "$DATASET_PERCENTAGE"
--bench_steps "$BENCH_STEPS"
--warmup_steps "$WARMUP_STEPS"
$ACTIVATION_CHECKPOINTING_FLAG
$SAVE_CHECKPOINT_ARG
$WANDB_FLAG
--wandb_project "$WANDB_PROJECT"
--wandb_run_name "$WANDB_RUN_NAME"
$DETERMINISTIC_FLAG
)

echo "Starting training with MODE $MODE"
echo "================================================"
"${CMD[@]}"

echo "================================================"
echo "Training completed"
echo "================================================"
Loading