Add Megatron data manifest and g0050 setup

This commit is contained in:
2026-07-02 20:50:24 +08:00
parent 816eccb5b5
commit 5609b1f8e4
7 changed files with 608 additions and 19 deletions
+29
View File
@@ -2,4 +2,33 @@
- `sync_pretrain_data_into_repo.sh`: 200B 数据构建完成后,把数据目录同步到 `dataset/pretrain/data/`,默认优先 hardlink。
- `wait_and_sync_pretrain_data.sh`: 后台等待当前 200B 构建进程结束,然后自动同步数据。
- `preprocess_megatron_bridge_pretrain.sh`: 旧的 Megatron indexed dataset 预处理入口,保留用于对照。
- `preprocess_megatron_bridge_pretrain_direct.sh`: 直接从 parquet 生成 Megatron indexed dataset,不落中间 JSONL。
- `train_megatron_bridge_2b_moe.sh`: 当前主训练入口,使用 NeMo 26.06 镜像中的 Megatron-Bridge。
- `train_nemo_megatron_2b_moe.sh`: NeMo/Megatron 训练入口占位,包含 image、mount、路径检查。
- `g0050_download_and_setup_from_modelscope.sh`: 在 g0050 上一键准备训练环境并从 ModelScope 下载未 tokenize parquet 数据。
## g0050 下载与部署
在 Mac 侧通过 B300 跳转到 g0050:
```bash
ssh B300 'ssh ubuntu@g0050 "cd /ssd/workspace/yi/laoyao_2b_moe && MODELSCOPE_API_TOKEN=ms-... bash scripts/g0050_download_and_setup_from_modelscope.sh"'
```
默认行为:
- repo 路径:`/ssd/workspace/yi/laoyao_2b_moe`
- 数据路径:`/ssd/workspace/yi/laoyao_2b_moe_pretraining_dataset`
- ModelScope dataset:`eigentom/laoyao_2b_moe_pretrain_parquet_20260702`
- 训练镜像:`nvcr.io/nvidia/nemo:26.06`
- 下载镜像:ModelScope CUDA 13.0 / Swift 4.3.1 官方镜像
- 代理:默认使用 B300/g0050 侧的 `http://100.72.0.101:8888`
私有 Gitea clone 时不要把 token 写进脚本,运行时通过环境变量传入:
```bash
GIT_REPO_URL=https://yi_lu:<token>@git.deeepseek.net/yi_lu/laoyao_2b_moe.git \
MODELSCOPE_API_TOKEN=ms-... \
bash scripts/g0050_download_and_setup_from_modelscope.sh
```
+92
View File
@@ -0,0 +1,92 @@
#!/usr/bin/env bash
set -euo pipefail
# Prepare g0050 for Laoyao 2B MoE training and download the pre-tokenization
# parquet dataset from ModelScope. Keep credentials out of this file:
#
# MODELSCOPE_API_TOKEN=ms-... bash scripts/g0050_download_and_setup_from_modelscope.sh
#
# Optional private git URL:
#
# GIT_REPO_URL=https://user:token@git.deeepseek.net/yi_lu/laoyao_2b_moe.git ...
REPO_ROOT="${REPO_ROOT:-/ssd/workspace/yi/laoyao_2b_moe}"
DATA_ROOT="${DATA_ROOT:-/ssd/workspace/yi/laoyao_2b_moe_pretraining_dataset}"
DATASET_ID="${DATASET_ID:-eigentom/laoyao_2b_moe_pretrain_parquet_20260702}"
GIT_REPO_URL="${GIT_REPO_URL:-https://git.deeepseek.net/yi_lu/laoyao_2b_moe.git}"
GIT_BRANCH="${GIT_BRANCH:-main}"
NEMO_IMAGE="${NEMO_IMAGE:-nvcr.io/nvidia/nemo:26.06}"
MODELSCOPE_IMAGE="${MODELSCOPE_IMAGE:-modelscope-registry.cn-hangzhou.cr.aliyuncs.com/modelscope-repo/modelscope:ubuntu22.04-cuda13.0.3-py312-torch2.11.0-vllm0.23.0-modelscope1.37.1-swift4.3.1}"
MODELSCOPE_ENDPOINT="${MODELSCOPE_ENDPOINT:-https://www.modelscope.cn}"
MAX_WORKERS="${MAX_WORKERS:-8}"
HTTP_PROXY_DEFAULT="${HTTP_PROXY_DEFAULT:-http://100.72.0.101:8888}"
HTTPS_PROXY_DEFAULT="${HTTPS_PROXY_DEFAULT:-http://100.72.0.101:8888}"
export http_proxy="${http_proxy:-$HTTP_PROXY_DEFAULT}"
export https_proxy="${https_proxy:-$HTTPS_PROXY_DEFAULT}"
export HTTP_PROXY="${HTTP_PROXY:-$http_proxy}"
export HTTPS_PROXY="${HTTPS_PROXY:-$https_proxy}"
if [[ -z "${MODELSCOPE_API_TOKEN:-}" ]]; then
echo "ERROR: set MODELSCOPE_API_TOKEN before running this script." >&2
exit 2
fi
mkdir -p "$(dirname "$REPO_ROOT")" "$DATA_ROOT"
echo "[setup] repo_root=$REPO_ROOT"
if [[ -d "$REPO_ROOT/.git" ]]; then
git -C "$REPO_ROOT" fetch origin "$GIT_BRANCH"
git -C "$REPO_ROOT" checkout "$GIT_BRANCH"
git -C "$REPO_ROOT" pull --ff-only origin "$GIT_BRANCH"
elif [[ -d "$REPO_ROOT" ]] && [[ -z "$(find "$REPO_ROOT" -mindepth 1 -maxdepth 1 -print -quit)" ]]; then
rmdir "$REPO_ROOT"
git clone --branch "$GIT_BRANCH" "$GIT_REPO_URL" "$REPO_ROOT"
else
git clone --branch "$GIT_BRANCH" "$GIT_REPO_URL" "$REPO_ROOT"
fi
echo "[setup] checking docker images"
if ! docker image inspect "$NEMO_IMAGE" >/dev/null 2>&1; then
docker pull "$NEMO_IMAGE"
fi
if ! docker image inspect "$MODELSCOPE_IMAGE" >/dev/null 2>&1; then
docker pull "$MODELSCOPE_IMAGE"
fi
echo "[download] dataset=$DATASET_ID"
echo "[download] data_root=$DATA_ROOT"
docker run --rm \
--network=host \
-e http_proxy="$http_proxy" \
-e https_proxy="$https_proxy" \
-e HTTP_PROXY="$HTTP_PROXY" \
-e HTTPS_PROXY="$HTTPS_PROXY" \
-e MODELSCOPE_API_TOKEN="$MODELSCOPE_API_TOKEN" \
-v "$DATA_ROOT:$DATA_ROOT" \
"$MODELSCOPE_IMAGE" \
modelscope download \
--dataset "$DATASET_ID" \
--token "$MODELSCOPE_API_TOKEN" \
--endpoint "$MODELSCOPE_ENDPOINT" \
--local_dir "$DATA_ROOT" \
--max-workers "$MAX_WORKERS"
echo "[verify] downloaded tree"
find "$DATA_ROOT" -maxdepth 3 -type f | sed -n '1,20p'
du -sh "$DATA_ROOT" || true
cat <<EOF
Done.
Expected source parquet roots:
$DATA_ROOT/train/pretrain_rebalanced_web40_edu20_chinese10_science10_logic10_math5_code5_200b_v1_20260701
$DATA_ROOT/train/logic_topup_proof_pile_17b_v1_20260701
Next steps:
cd $REPO_ROOT
bash scripts/preprocess_megatron_bridge_pretrain_direct.sh
bash scripts/train_megatron_bridge_2b_moe.sh
EOF
+34 -11
View File
@@ -3,36 +3,59 @@ set -euo pipefail
REPO_ROOT="${REPO_ROOT:-/mnt/beegfs/yi/laoyao_2b_moe}"
IMAGE="${IMAGE:-nvcr.io/nvidia/nemo:26.06}"
SOURCE_DATA="${SOURCE_DATA:-/mnt/beegfs/yi/laoyao_2b_moe_pretraining_dataset/train/pretrain_rebalanced_web40_edu20_chinese10_science10_logic10_math5_code5_200b_v1_20260701}"
SOURCE_DATA_DIRS="${SOURCE_DATA_DIRS:-/mnt/beegfs/yi/laoyao_2b_moe_pretraining_dataset/train/pretrain_rebalanced_web40_edu20_chinese10_science10_logic10_math5_code5_200b_v1_20260701:/mnt/beegfs/yi/laoyao_2b_moe_pretraining_dataset/train/logic_topup_proof_pile_17b_v1_20260701}"
WORK_DIR="${WORK_DIR:-/mnt/beegfs/yi/laoyao_2b_moe_pretraining_dataset/megatron_bridge/pretrain_8192_v1}"
JSONL="${JSONL:-$WORK_DIR/text.jsonl}"
OUTPUT_PREFIX="${OUTPUT_PREFIX:-$WORK_DIR/laoyao_2b_moe_8192_text_document}"
TOKENIZER_MODEL="${TOKENIZER_MODEL:-$REPO_ROOT/tokenizer/glm5.2}"
WORKERS="${WORKERS:-16}"
MAX_DOCS="${MAX_DOCS:-0}"
MIN_FREE_GB="${MIN_FREE_GB:-1000}"
KEEP_JSONL="${KEEP_JSONL:-0}"
mkdir -p "$WORK_DIR"
available_gb="$(df -BG "$WORK_DIR" | awk 'NR==2 {gsub("G", "", $4); print $4}')"
if [[ "$MAX_DOCS" == "0" && "$available_gb" -lt "$MIN_FREE_GB" ]]; then
cat >&2 <<EOF
Refusing full preprocessing: only ${available_gb}GB free at $WORK_DIR.
Full 200B-token Megatron indexed output is expected to require hundreds of GB,
and this script also stages a JSONL export. Set MIN_FREE_GB lower only if you
have confirmed an output filesystem with enough capacity, or run a bounded
probe with MAX_DOCS.
EOF
exit 2
fi
input_args=()
IFS=':' read -r -a source_dirs <<< "$SOURCE_DATA_DIRS"
for source_dir in "${source_dirs[@]}"; do
input_args+=(--input "${source_dir}/*.parquet")
done
docker run --rm --ipc=host --network=host \
--ulimit memlock=-1 --ulimit stack=67108864 \
-v /mnt/beegfs:/mnt/beegfs \
-w "$REPO_ROOT" \
"$IMAGE" \
bash -lc "
bash -lc '
set -euo pipefail
python3 dataset/pretrain/scripts/export_pretrain_parquet_text_jsonl.py \
--input '$SOURCE_DATA/*.parquet' \
--output '$JSONL' \
--max-docs '$MAX_DOCS'
"$@" \
--output '"$JSONL"' \
--max-docs '"$MAX_DOCS"'
python3 /opt/Megatron-Bridge/3rdparty/Megatron-LM/tools/preprocess_data.py \
--input '$JSONL' \
--input '"$JSONL"' \
--json-keys text \
--tokenizer-type HuggingFaceTokenizer \
--tokenizer-model '$TOKENIZER_MODEL' \
--tokenizer-model '"$TOKENIZER_MODEL"' \
--append-eod \
--output-prefix '$OUTPUT_PREFIX' \
--workers '$WORKERS'
ls -lh '${OUTPUT_PREFIX}'*
"
--output-prefix '"$OUTPUT_PREFIX"' \
--workers '"$WORKERS"'
if [[ "'"$KEEP_JSONL"'" == "0" ]]; then
rm -f '"$JSONL"'
fi
ls -lh '"${OUTPUT_PREFIX}"'*
' bash "${input_args[@]}"
echo "Megatron indexed dataset prefix: $OUTPUT_PREFIX"
+72
View File
@@ -0,0 +1,72 @@
#!/usr/bin/env bash
set -euo pipefail
REPO_ROOT="${REPO_ROOT:-/mnt/beegfs/yi/laoyao_2b_moe}"
IMAGE="${IMAGE:-nvcr.io/nvidia/nemo:26.06}"
SOURCE_DATA_DIRS="${SOURCE_DATA_DIRS:-/mnt/beegfs/yi/laoyao_2b_moe_pretraining_dataset/train/pretrain_rebalanced_web40_edu20_chinese10_science10_logic10_math5_code5_200b_v1_20260701:/mnt/beegfs/yi/laoyao_2b_moe_pretraining_dataset/train/logic_topup_proof_pile_17b_v1_20260701}"
WORK_DIR="${WORK_DIR:-/mnt/beegfs/yi/laoyao_2b_moe_pretraining_dataset/megatron_bridge/pretrain_8192_direct_smoke_v1}"
TOKENIZER_MODEL="${TOKENIZER_MODEL:-$REPO_ROOT/tokenizer/glm5.2}"
OUTPUT_PREFIX_PREFIX="${OUTPUT_PREFIX_PREFIX:-laoyao_2b_moe_8192_direct}"
TEXT_KEY="${TEXT_KEY:-text}"
PARALLEL_FILES="${PARALLEL_FILES:-1}"
WORKERS_PER_FILE="${WORKERS_PER_FILE:-8}"
BATCH_SIZE="${BATCH_SIZE:-8192}"
CHUNKSIZE="${CHUNKSIZE:-128}"
MAX_FILES="${MAX_FILES:-1}"
MAX_DOCS="${MAX_DOCS:-100000}"
MAX_SEQ_LEN="${MAX_SEQ_LEN:-65536}"
MIN_FREE_GB="${MIN_FREE_GB:-20}"
OVERWRITE="${OVERWRITE:-1}"
mkdir -p "$WORK_DIR"
available_gb="$(df -BG "$WORK_DIR" | awk 'NR==2 {gsub("G", "", $4); print $4}')"
if [[ "$MAX_FILES" == "0" && "$MAX_DOCS" == "0" && "$available_gb" -lt "$MIN_FREE_GB" ]]; then
cat >&2 <<EOF
Refusing full direct preprocessing: only ${available_gb}GB free at $WORK_DIR.
Direct Megatron indexed output for 200B GLM5.2 tokens is expected to require
roughly 800GB plus index/cache overhead. Increase free space or lower MAX_FILES/MAX_DOCS.
EOF
exit 2
fi
input_args=()
IFS=':' read -r -a source_dirs <<< "$SOURCE_DATA_DIRS"
for source_dir in "${source_dirs[@]}"; do
input_args+=(--input "${source_dir}/*.parquet")
done
max_file_args=()
if [[ "$MAX_FILES" != "0" ]]; then
max_file_args+=(--max-files "$MAX_FILES")
fi
overwrite_args=()
if [[ "$OVERWRITE" == "1" ]]; then
overwrite_args+=(--overwrite)
fi
docker run --rm --ipc=host --network=host \
--ulimit memlock=-1 --ulimit stack=67108864 \
-v /mnt/beegfs:/mnt/beegfs \
-w "$REPO_ROOT" \
"$IMAGE" \
python3 dataset/pretrain/scripts/convert_pretrain_parquet_to_megatron.py \
"${input_args[@]}" \
--output-dir "$WORK_DIR" \
--manifest "$WORK_DIR/manifest.json" \
--megatron-dir /opt/Megatron-Bridge/3rdparty/Megatron-LM \
--tokenizer-type HuggingFaceTokenizer \
--tokenizer-model "$TOKENIZER_MODEL" \
--text-key "$TEXT_KEY" \
--output-prefix-prefix "$OUTPUT_PREFIX_PREFIX" \
--parallel-files "$PARALLEL_FILES" \
--workers-per-file "$WORKERS_PER_FILE" \
--batch-size "$BATCH_SIZE" \
--chunksize "$CHUNKSIZE" \
--max-docs "$MAX_DOCS" \
--max-seq-len "$MAX_SEQ_LEN" \
"${max_file_args[@]}" \
"${overwrite_args[@]}"
echo "Direct Megatron indexed dataset manifest: $WORK_DIR/manifest.json"
+19 -5
View File
@@ -4,7 +4,8 @@ set -euo pipefail
REPO_ROOT="${REPO_ROOT:-/mnt/beegfs/yi/laoyao_2b_moe}"
IMAGE="${IMAGE:-nvcr.io/nvidia/nemo:26.06}"
NPROC_PER_NODE="${NPROC_PER_NODE:-8}"
DATA_PREFIX="${DATA_PREFIX:-/mnt/beegfs/yi/laoyao_2b_moe_pretraining_dataset/megatron_bridge/pretrain_8192_v1/laoyao_2b_moe_8192_text_document}"
DATA_PREFIX="${DATA_PREFIX:-}"
DATA_MANIFEST="${DATA_MANIFEST:-/mnt/beegfs/yi/laoyao_2b_moe_pretraining_dataset/megatron_bridge/pretrain_65536_direct_v1/manifest.json}"
TRAIN_ITERS="${TRAIN_ITERS:-10}"
SEQ_LENGTH="${SEQ_LENGTH:-8192}"
MICRO_BATCH_SIZE="${MICRO_BATCH_SIZE:-1}"
@@ -15,9 +16,15 @@ EP="${EP:-1}"
CP="${CP:-1}"
DRY_RUN="${DRY_RUN:-0}"
if [[ "$DRY_RUN" != "1" && ! -f "${DATA_PREFIX}.idx" ]]; then
echo "missing Megatron indexed data prefix: $DATA_PREFIX" >&2
echo "run scripts/preprocess_megatron_bridge_pretrain.sh first, or set DATA_PREFIX" >&2
if [[ -n "$DATA_MANIFEST" ]]; then
if [[ ! -f "$DATA_MANIFEST" ]]; then
echo "missing Megatron indexed dataset manifest: $DATA_MANIFEST" >&2
echo "run scripts/preprocess_megatron_bridge_pretrain_direct.sh first, or set DATA_MANIFEST" >&2
exit 1
fi
elif [[ "$DRY_RUN" != "1" && ( -z "$DATA_PREFIX" || ! -f "${DATA_PREFIX}.idx" ) ]]; then
echo "missing Megatron indexed data prefix: ${DATA_PREFIX:-<empty>}" >&2
echo "set DATA_MANIFEST or DATA_PREFIX" >&2
exit 1
fi
@@ -26,6 +33,13 @@ if [[ "$DRY_RUN" == "1" ]]; then
DRY_RUN_ARG="--dry-run"
fi
DATA_ARGS=()
if [[ -n "$DATA_MANIFEST" ]]; then
DATA_ARGS=(--data-manifest "$DATA_MANIFEST")
else
DATA_ARGS=(--data-prefix "$DATA_PREFIX")
fi
docker run --rm --gpus all --ipc=host --network=host \
--ulimit memlock=-1 --ulimit stack=67108864 \
-v /mnt/beegfs:/mnt/beegfs \
@@ -35,7 +49,7 @@ docker run --rm --gpus all --ipc=host --network=host \
set -euo pipefail
torchrun --nproc_per_node='$NPROC_PER_NODE' \
training/megatron_bridge/laoyao_2b_moe_pretrain.py \
--data-prefix '$DATA_PREFIX' \
${DATA_ARGS[*]} \
--seq-length '$SEQ_LENGTH' \
--train-iters '$TRAIN_ITERS' \
--micro-batch-size '$MICRO_BATCH_SIZE' \