From 4057878c3486068f3d73c5659ff1982ef503dc52 Mon Sep 17 00:00:00 2001 From: wangxiaoxin-sherie Date: Fri, 14 Aug 2026 10:38:08 +0800 Subject: [PATCH 01/12] Apply-MegatronAdaptor-NPU-migration-to-clean-branch Signed-off-by: wangxiaoxin-sherie --- docker/Dockerfile.npu | 20 +- docker/npu_patch/README.md | 22 +- docker/npu_patch/megatron.patch | 262 ++++-- docker/npu_patch/megatron_comm.patch | 772 ------------------ docker/npu_patch/mindspeed.patch | 280 ------- docker/npu_patch/series.conf | 2 - docker/patch/latest/vllm.patch | 2 +- examples/retool/retool_qwen3_4b_rl.sh | 4 +- examples/retool/retool_qwen3_4b_sft.sh | 4 +- examples/search-r1/run_qwen3_4b_npu.sh | 4 +- .../test_update_weight_from_tensor.py | 2 +- train.py | 2 +- train_async.py | 2 +- vime/backends/megatron_utils/__init__.py | 5 +- vime/backends/megatron_utils/actor.py | 16 +- .../megatron_utils/npu_attention_patch.py | 5 +- .../update_weight_from_tensor.py | 2 +- vime/utils/arguments.py | 5 +- vime/utils/external_utils/launch.py | 3 +- 19 files changed, 237 insertions(+), 1177 deletions(-) delete mode 100644 docker/npu_patch/megatron_comm.patch delete mode 100644 docker/npu_patch/mindspeed.patch diff --git a/docker/Dockerfile.npu b/docker/Dockerfile.npu index 17d9d6782..c2bdead55 100644 --- a/docker/Dockerfile.npu +++ b/docker/Dockerfile.npu @@ -7,9 +7,8 @@ FROM ${BASE_IMAGE}:${BASE_IMAGE_TAG} SHELL ["/bin/bash", "-o", "pipefail", "-c"] WORKDIR /root -ARG MEGATRON_COMMIT=3714d81d418c9f1bca4594fc35f9e8289f652862 +ARG MEGATRON_COMMIT=1dcf0dafa884ad52ffb243625717a3471643e087 ARG MEGATRON_BRIDGE_COMMIT=3fd3768045422d0aa5c97e90a4e6c659aea9acb9 -ARG MINDSPEED_COMMIT=fc63de5c48426dd019c3b3f39e65f5bdf56e4086 ARG MBRIDGE_COMMIT=89eb10887887bc74853f89a4de258c0702932a1c ARG SOC_VERSION="ascend910_9391" ARG PIP_INDEX_URL="https://mirrors.tuna.tsinghua.edu.cn/pypi/web/simple" @@ -28,7 +27,7 @@ ENV SOC_VERSION=$SOC_VERSION \ HCCL_NPU_SOCKET_PORT_RANGE=61000-61050 \ PYTORCH_NPU_ALLOC_CONF=expandable_segments:True \ HYDRA_FULL_ERROR=1 \ - PYTHONPATH=/root/Megatron-Bridge/src:/root/Megatron-LM:/root/vime + PYTHONPATH=/root/Megatron-Bridge/src:/root/Megatron-LM:/root/MegatronAdaptor:/root/TransformerEngineNPU:/root/vime # PATCH MAINTENANCE: keep patch COPY/apply operations and # docker/npu_patch/series.conf synchronized. @@ -72,28 +71,25 @@ RUN git clone --branch bridge https://github.com/radixark/Megatron-Bridge.git \ /root/Megatron-Bridge && \ git -C /root/Megatron-Bridge checkout "${MEGATRON_BRIDGE_COMMIT}" -RUN git clone https://gitcode.com/Ascend/MindSpeed.git /root/MindSpeed && \ - git -C /root/MindSpeed checkout "${MINDSPEED_COMMIT}" +RUN git clone https://gitcode.com/Ascend/MegatronAdaptor.git /root/MegatronAdaptor && \ + git clone https://gitcode.com/Ascend/TransformerEngineNPU.git /root/TransformerEngineNPU RUN git clone https://github.com/ISEEKYAN/mbridge.git /root/mbridge && \ git -C /root/mbridge checkout "${MBRIDGE_COMMIT}" # Apply NPU training-stack patches from the build-context snapshot. RUN git -C /root/Megatron-LM apply --whitespace=nowarn \ - /opt/npu_patch/megatron_comm.patch && \ - git -C /root/Megatron-LM apply --whitespace=nowarn \ /opt/npu_patch/megatron.patch && \ git -C /root/Megatron-Bridge apply --whitespace=nowarn \ - /opt/npu_patch/megatron-bridge.patch && \ - git -C /root/MindSpeed apply --whitespace=nowarn \ - /opt/npu_patch/mindspeed.patch + /opt/npu_patch/megatron-bridge.patch # Megatron-Bridge is used directly from PYTHONPATH. Installing its package # metadata would pull CUDA-only dependencies into the Ascend environment. RUN pip install --no-build-isolation "nvidia-modelopt[torch]>=0.37.0" && \ pip install --no-deps --no-build-isolation -e /root/mbridge && \ pip install --no-deps --no-build-isolation -e /root/Megatron-LM && \ - pip install --no-deps --no-build-isolation -e /root/MindSpeed + pip install --no-deps --no-build-isolation -e /root/TransformerEngineNPU && \ + pip install --no-deps --no-build-isolation -e /root/MegatronAdaptor # Defaults to the ascend branch for local builds. Release workflows should # pass an immutable commit SHA for reproducible images. @@ -123,7 +119,7 @@ RUN git clone --depth 1 --branch 2026.6.0 \ # Minimal import check. RUN source /usr/local/Ascend/ascend-toolkit/set_env.sh && \ - python3 -c 'import megatron, mindspeed, torch_memory_saver, vime, vllm, vllm_ascend;' + python3 -c 'import megatron, megatron_adaptor, transformer_engine, torch_memory_saver, vime, vllm, vllm_ascend;' WORKDIR /root/vime ENTRYPOINT [] diff --git a/docker/npu_patch/README.md b/docker/npu_patch/README.md index 8f376aec2..19caa4a74 100644 --- a/docker/npu_patch/README.md +++ b/docker/npu_patch/README.md @@ -8,8 +8,9 @@ This guide provides instructions for installing Vime with NPU support, including | --------------- | ---------------------------------------- | ------------------------------------------------------------------------------------------------------------------- | | vime | main | [GitHub](https://github.com/vllm-project/vime/tree/main) | | Megatron-Bridge | 3fd3768045422d0aa5c97e90a4e6c659aea9acb9 | [GitHub](https://github.com/radixark/Megatron-Bridge) | -| Megatron-LM | 3714d81d418c9f1bca4594fc35f9e8289f652862 | [GitHub](https://github.com/NVIDIA/Megatron-LM) | -| MindSpeed | fc63de5c48426dd019c3b3f39e65f5bdf56e4086 | [GitCode](https://gitcode.com/Ascend/MindSpeed) | +| Megatron-LM | 1dcf0dafa884ad52ffb243625717a3471643e087 | [GitHub](https://github.com/NVIDIA/Megatron-LM) | +| MegatronAdaptor | main | [GitCode](https://gitcode.com/Ascend/MegatronAdaptor) | +| TransformerEngineNPU | main | [GitCode](https://gitcode.com/Ascend/TransformerEngineNPU) | | HDK | 25.3.RC1 | [Ascend](https://www.hiascend.com/hardware/firmware-drivers/commercial?product=7\&model=33) | | CANN | 9.0.0 | [Ascend](https://www.hiascend.com/developer/download/community/result?module=cann\&cann=9.0.0\&product=7\&model=33) | @@ -52,27 +53,24 @@ pip install --no-build-isolation "nvidia-modelopt[torch]>=0.37.0" #### 2. Megatron-LM ```bash -export MEGATRON_COMMIT=3714d81d418c9f1bca4594fc35f9e8289f652862 +export MEGATRON_COMMIT=1dcf0dafa884ad52ffb243625717a3471643e087 git clone https://github.com/NVIDIA/Megatron-LM.git "${WORKSPACE}/Megatron-LM" git -C "${WORKSPACE}/Megatron-LM" checkout "${MEGATRON_COMMIT}" -git -C "${WORKSPACE}/Megatron-LM" apply --whitespace=nowarn "${PATCH_DIR}/megatron_comm.patch" +git -C "${WORKSPACE}/Megatron-LM" apply --whitespace=nowarn "${WORKSPACE}/vime/docker/patch/latest/megatron.patch" git -C "${WORKSPACE}/Megatron-LM" apply --whitespace=nowarn "${PATCH_DIR}/megatron.patch" pip install --no-deps --no-build-isolation -e "${WORKSPACE}/Megatron-LM" ``` -#### 3. MindSpeed +#### 3. MegatronAdaptor and TransformerEngineNPU -```bash -export MINDSPEED_COMMIT=fc63de5c48426dd019c3b3f39e65f5bdf56e4086 -git clone https://gitcode.com/Ascend/MindSpeed.git "${WORKSPACE}/MindSpeed" -git -C "${WORKSPACE}/MindSpeed" checkout "${MINDSPEED_COMMIT}" +The NPU training stack now uses the two source repositories directly. The mainline Megatron patch is applied first; `docker/npu_patch/megatron.patch` contains only the NPU-specific changes rebased onto that mainline patch: -git -C "${WORKSPACE}/MindSpeed" apply --whitespace=nowarn "${PATCH_DIR}/mindspeed.patch" +pip install --no-deps --no-build-isolation -e ${WORKSPACE}/MegatronAdaptor +pip install --no-deps --no-build-isolation -e ${WORKSPACE}/TransformerEngineNPU -pip install --no-deps --no-build-isolation -e "${WORKSPACE}/MindSpeed" -``` +Do not install the CUDA TransformerEngine package in the same environment. #### 4. Vime diff --git a/docker/npu_patch/megatron.patch b/docker/npu_patch/megatron.patch index 296947a7e..53eb510ed 100644 --- a/docker/npu_patch/megatron.patch +++ b/docker/npu_patch/megatron.patch @@ -25,6 +25,92 @@ index 8b422d73a..58fba4667 100644 def fast_gelu(x: torch.Tensor) -> torch.Tensor: """Fast GELU activation""" return 0.5 * x * (1.0 + torch.tanh(x * 0.7978845608 * (1.0 + 0.044715 * x * x))) +diff --git a/megatron/core/distributed/param_and_grad_buffer.py b/megatron/core/distributed/param_and_grad_buffer.py +index a9982e176..807c4a88d 100644 +--- a/megatron/core/distributed/param_and_grad_buffer.py ++++ b/megatron/core/distributed/param_and_grad_buffer.py +@@ -787,45 +787,19 @@ class _ParamAndGradBuffer: + # Individual param/grad contexts below handle TMS regions separately. + mem_alloc_context = nullcontext + +- def _make_no_backup_context(tag, disable, flag_name="disable_grad_buffers_cpu_backup"): +- if disable: +- try: +- from torch_memory_saver import torch_memory_saver +- except ImportError as e: +- raise ImportError( +- f"{flag_name}=True requires torch_memory_saver. " +- "Install with: pip install torch-memory-saver" +- ) from e +- return partial( +- torch_memory_saver.region, +- tag=tag, +- enable_cpu_backup=False, +- ) +- return nullcontext +- grad_mem_alloc_context = _make_no_backup_context( +- "grad_buffer", disable_grad_buffers_cpu_backup +- ) +- param_mem_alloc_context = _make_no_backup_context( +- "param_buffer", disable_param_buffers_cpu_backup, "disable_param_buffers_cpu_backup" +- ) +- ++ # NPU already wraps model construction in the outer VIME training ++ # torch_memory_saver region. Do not open nested mem-pool regions here. + with mem_alloc_context(): + # For MXFP8 param: Create a shared buffer for param AG and grad RS for memory efficiency + # The buffer is mapped to weight gradients whose dtype is either bf16 or FP32. + # It can be temporarily reused by param AG. + if self.ddp_config.use_distributed_optimizer and any(is_mxfp8tensor(p) for p in params): +- shared_mem_alloc_context = ( +- param_mem_alloc_context +- if disable_param_buffers_cpu_backup +- else grad_mem_alloc_context ++ self.shared_buffer = torch.zeros( ++ self.numel, ++ dtype=self.grad_dtype, ++ device=torch.cuda.current_device(), ++ requires_grad=False, + ) +- with shared_mem_alloc_context(): +- self.shared_buffer = torch.zeros( +- self.numel, +- dtype=self.grad_dtype, +- device=torch.cuda.current_device(), +- requires_grad=False, +- ) + # For FP32 weight grads, only half of the buffer is used to store params in bf16. + if self.grad_dtype == torch.float32: + self.param_data = self.shared_buffer[: math.ceil(self.numel / 2)].view( +@@ -837,20 +811,18 @@ class _ParamAndGradBuffer: + else: + # Only re-map param tensors if using distributed optimizer. + if self.ddp_config.use_distributed_optimizer: +- with param_mem_alloc_context(): +- self.param_data = torch.zeros( +- self.numel, +- dtype=self.param_dtype, +- device=torch.cuda.current_device(), +- requires_grad=False, +- ) +- with grad_mem_alloc_context(): +- self.grad_data = torch.zeros( ++ self.param_data = torch.zeros( + self.numel, +- dtype=self.grad_dtype, ++ dtype=self.param_dtype, + device=torch.cuda.current_device(), + requires_grad=False, + ) ++ self.grad_data = torch.zeros( ++ self.numel, ++ dtype=self.grad_dtype, ++ device=torch.cuda.current_device(), ++ requires_grad=False, ++ ) + + self.grad_data_size = 0 + self.param_data_size = 0 diff --git a/megatron/core/fusions/fused_bias_dropout.py b/megatron/core/fusions/fused_bias_dropout.py index 336452562..614ee1a48 100644 --- a/megatron/core/fusions/fused_bias_dropout.py @@ -311,11 +397,29 @@ index 02dabc14c..137022386 100644 def weighted_squared_relu_back(g: torch.Tensor, x: torch.Tensor, weights: torch.Tensor): """Backward for weighted Squared-ReLU. +diff --git a/megatron/core/inference/contexts/dynamic_context.py b/megatron/core/inference/contexts/dynamic_context.py +index 2b241c86d..cca4503d5 100644 +--- a/megatron/core/inference/contexts/dynamic_context.py ++++ b/megatron/core/inference/contexts/dynamic_context.py +@@ -51,12 +51,8 @@ try: + except: + HAVE_PACKAGING = False + +-try: +- import flashinfer # pylint: disable=unused-import + +- HAVE_FLASHINFER = True +-except ImportError: +- HAVE_FLASHINFER = False ++HAVE_FLASHINFER = False + + try: + from torch_memory_saver import torch_memory_saver diff --git a/megatron/core/models/gpt/gpt_layer_specs.py b/megatron/core/models/gpt/gpt_layer_specs.py -index 712793853..76ea4333b 100755 +index a9d3583e5..f2adaceb2 100755 --- a/megatron/core/models/gpt/gpt_layer_specs.py +++ b/megatron/core/models/gpt/gpt_layer_specs.py -@@ -219,7 +219,7 @@ def get_gpt_layer_with_transformer_engine_spec( +@@ -209,7 +209,7 @@ def get_gpt_layer_with_transformer_engine_spec( 'The fp8 argument in "get_gpt_layer_with_transformer_engine_spec" has been deprecated' " and will be removed soon. Please update your code accordingly." ) @@ -324,11 +428,38 @@ index 712793853..76ea4333b 100755 if use_kitchen: assert HAVE_KITCHEN backend: BackendSpecProvider = KitchenSpecProvider( +diff --git a/megatron/core/optimizer/distrib_optimizer.py b/megatron/core/optimizer/distrib_optimizer.py +index dd8590a79..3736b0e31 100644 +--- a/megatron/core/optimizer/distrib_optimizer.py ++++ b/megatron/core/optimizer/distrib_optimizer.py +@@ -659,10 +659,12 @@ class DistributedOptimizer(MixedPrecisionOptimizer): + break + elif USING_TE_OPTIMIZER or USING_APEX_OPTIMIZER: + # Extract 'step', for TE FusedAdam support. ++ # Use int() to convert tensors to Python scalars so set() deduplicates ++ # by value (PyTorch tensor hashing is identity-based). + steps = list( + set( + [ +- g["step"] ++ int(g["step"]) + for g in inner_state_dict["param_groups"] + if len(g["params"]) > 0 and "step" in g + ] +@@ -824,7 +826,7 @@ class DistributedOptimizer(MixedPrecisionOptimizer): + + # Extract 'step', for non-Apex/TE support. + if not HAVE_APEX_OR_TE: +- steps = list(set([g["step"] for g in state_dict["optimizer"]["param_groups"]])) ++ steps = list(set([int(g["step"]) for g in state_dict["optimizer"]["param_groups"]])) + assert len(steps) == 1 + step = torch.tensor(steps[0], dtype=torch.float) + diff --git a/megatron/core/ssm/gated_delta_net.py b/megatron/core/ssm/gated_delta_net.py -index dfa6e4c35..8b3621819 100644 +index 601a72a43..ec114cf38 100644 --- a/megatron/core/ssm/gated_delta_net.py +++ b/megatron/core/ssm/gated_delta_net.py -@@ -413,7 +413,7 @@ class GatedDeltaNet(MegatronModule): +@@ -465,7 +465,7 @@ class GatedDeltaNet(MegatronModule): return out, out_bias @@ -338,10 +469,10 @@ index dfa6e4c35..8b3621819 100644 # Output Norm x_dtype = x.dtype diff --git a/megatron/core/transformer/attention.py b/megatron/core/transformer/attention.py -index 80e9ec6fc..b6bcef6a9 100644 +index bc5e4e2ee..d223e774e 100644 --- a/megatron/core/transformer/attention.py +++ b/megatron/core/transformer/attention.py -@@ -1026,7 +1026,7 @@ class Attention(MegatronModule, ABC): +@@ -1219,7 +1219,7 @@ class Attention(MegatronModule, ABC): return output, bias @@ -351,10 +482,10 @@ index 80e9ec6fc..b6bcef6a9 100644 x_dtype = x.dtype gate = gate.contiguous() diff --git a/megatron/core/transformer/module.py b/megatron/core/transformer/module.py -index 2330df91b..0446c9097 100644 +index c30c107e7..f4340f4d1 100644 --- a/megatron/core/transformer/module.py +++ b/megatron/core/transformer/module.py -@@ -16,9 +16,9 @@ from megatron.core.transformer.utils import ( +@@ -17,9 +17,9 @@ from megatron.core.transformer.utils import ( sharded_state_dict_default, ) @@ -368,10 +499,10 @@ index 2330df91b..0446c9097 100644 def param_is_not_shared(param): # pylint: disable=missing-function-docstring diff --git a/megatron/core/transformer/moe/experts.py b/megatron/core/transformer/moe/experts.py -index 5eeafdd8d..a4dce6970 100644 +index d8e753422..aff929985 100644 --- a/megatron/core/transformer/moe/experts.py +++ b/megatron/core/transformer/moe/experts.py -@@ -91,7 +91,7 @@ class GroupedMLP(MegatronModule): +@@ -92,7 +92,7 @@ class GroupedMLP(MegatronModule): if self.config.activation_func not in (F.silu, F.gelu): raise ValueError("Activation function must be silu or gelu when using GroupedMLP.") @@ -380,7 +511,7 @@ index 5eeafdd8d..a4dce6970 100644 def glu(x): x = torch.chunk(x, 2, dim=-1) return self.config.activation_func(x[0]) * x[1] -@@ -108,7 +108,7 @@ class GroupedMLP(MegatronModule): +@@ -109,7 +109,7 @@ class GroupedMLP(MegatronModule): "moe_act recompute for fp8 or fp4 cannot work with the legacy GroupedMLP." ) @@ -389,24 +520,11 @@ index 5eeafdd8d..a4dce6970 100644 def activation_func_with_probs(x, probs): dtype = x.dtype res = self.activation_func(x) * probs -diff --git a/megatron/core/transformer/moe/router.py b/megatron/core/transformer/moe/router.py -index 517944f25..2c5bc0728 100644 ---- a/megatron/core/transformer/moe/router.py -+++ b/megatron/core/transformer/moe/router.py -@@ -472,7 +472,7 @@ class TopKRouter(Router): - else: - return input - -- @jit_fuser -+ - def _apply_expert_bias(self, routing_map: torch.Tensor): - """ - Update expert bias and tokens_per_expert diff --git a/megatron/core/transformer/moe/token_dispatcher.py b/megatron/core/transformer/moe/token_dispatcher.py -index d0da38d63..6d092c88f 100644 +index 327dbc8a3..b007d6662 100644 --- a/megatron/core/transformer/moe/token_dispatcher.py +++ b/megatron/core/transformer/moe/token_dispatcher.py -@@ -1403,7 +1403,7 @@ class MoEFlexTokenDispatcher(MoETokenDispatcher): +@@ -1402,7 +1402,7 @@ class MoEFlexTokenDispatcher(MoETokenDispatcher): ).contiguous() return routing_map, probs @@ -415,6 +533,18 @@ index d0da38d63..6d092c88f 100644 def dispatch_preprocess( self, hidden_states: torch.Tensor, routing_map: torch.Tensor, probs: torch.Tensor ): +diff --git a/megatron/core/transformer/multi_token_prediction.py b/megatron/core/transformer/multi_token_prediction.py +index 63f81465d..c47baeaea 100755 +--- a/megatron/core/transformer/multi_token_prediction.py ++++ b/megatron/core/transformer/multi_token_prediction.py +@@ -7,6 +7,7 @@ from typing import Callable, List, Optional, Union + + import torch + from torch import Tensor ++import warnings + + from megatron.core import InferenceParams, parallel_state, tensor_parallel + from megatron.core.dist_checkpointing.mapping import ShardedStateDict diff --git a/megatron/core/transformer/torch_norm.py b/megatron/core/transformer/torch_norm.py index d0ceca7af..f16796680 100644 --- a/megatron/core/transformer/torch_norm.py @@ -428,6 +558,27 @@ index d0ceca7af..f16796680 100644 def _norm(self, x): """ Performs the actual L2 normalization. +diff --git a/megatron/core/transformer/transformer_layer.py b/megatron/core/transformer/transformer_layer.py +index 227e95862..da4540185 100644 +--- a/megatron/core/transformer/transformer_layer.py ++++ b/megatron/core/transformer/transformer_layer.py +@@ -775,6 +775,16 @@ class TransformerLayer(GraphableMegatronModule, BaseTransformerLayer): + self._set_fc2_residual(residual) + mlp_output_with_bias = self.mlp(pre_mlp_layernorm_output, padding_mask=padding_mask) + ++ mlp_output, mlp_output_bias = mlp_output_with_bias ++ mlp_output = self.post_mlp_layernorm(mlp_output) ++ mlp_output_with_bias = (mlp_output, mlp_output_bias) ++ ++ if self.recompute_pre_mlp_layernorm: ++ # discard the output of the pre-mlp layernorm and register the recompute ++ # as a gradient hook of mlp_output_with_bias[0] ++ self.pre_mlp_norm_checkpoint.discard_output_and_register_recompute( ++ mlp_output_with_bias[0] ++ ) + nvtx_range_pop(suffix="mlp") + + if ( diff --git a/megatron/core/transformer/utils.py b/megatron/core/transformer/utils.py index 880c53099..dbc95736e 100644 --- a/megatron/core/transformer/utils.py @@ -473,7 +624,7 @@ index e00e63148..ffe4b7ec6 100644 x = bias + y tanh_out = torch.tanh(0.79788456 * x * (1 + 0.044715 * x * x)) diff --git a/megatron/legacy/model/transformer.py b/megatron/legacy/model/transformer.py -index 2a662a55b..2fc3e1bfb 100644 +index ca3414eec..5dd9e4798 100644 --- a/megatron/legacy/model/transformer.py +++ b/megatron/legacy/model/transformer.py @@ -856,7 +856,7 @@ def get_bias_dropout_add(training): @@ -515,33 +666,12 @@ index 5762000d5..534858df7 100644 + def erf_gelu(x): return x * 0.5 * (torch.erf(x / 1.41421).to(dtype=x.dtype)+torch.ones_like(x).to(dtype=x.dtype)) - - -diff --git a/megatron/core/inference/contexts/dynamic_context.py b/megatron/core/inference/contexts/dynamic_context.py -index 9b855effb..90c37c8db 100644 ---- a/megatron/core/inference/contexts/dynamic_context.py -+++ b/megatron/core/inference/contexts/dynamic_context.py -@@ -49,12 +49,8 @@ try: - except: - HAVE_PACKAGING = False - --try: -- import flashinfer # pylint: disable=unused-import -- HAVE_FLASHINFER = True --except ImportError: -- HAVE_FLASHINFER = False -+HAVE_FLASHINFER = False - - try: - import wandb # pylint: disable=unused-import - diff --git a/megatron/training/utils.py b/megatron/training/utils.py -index 95ad20382..c4d5e6f78 100644 +index 7709f6513..65b8f7fca 100644 --- a/megatron/training/utils.py +++ b/megatron/training/utils.py -@@ -445,6 +445,8 @@ def is_first_or_last_pipeline_stage(vp_stage): - +@@ -446,6 +446,8 @@ def is_first_or_last_pipeline_stage(vp_stage): def get_device_arch_version(): """Returns GPU arch version (8: Ampere, 9: Hopper, 10: Blackwell, ...)""" @@ -550,35 +680,3 @@ index 95ad20382..c4d5e6f78 100644 return torch.cuda.get_device_properties(torch.device("cuda:0")).major - def append_to_progress_log(string, barrier=True): - -diff --git a/megatron/core/dist_checkpointing/validation.py b/megatron/core/dist_checkpointing/validation.py -index 48f2bda87..028032cd8 100644 ---- a/megatron/core/dist_checkpointing/validation.py -+++ b/megatron/core/dist_checkpointing/validation.py -@@ -440,6 +440,8 @@ def validate_sharding_integrity( - for key, shardings in key_shardings.items(): - if isinstance(shardings[0][1], ShardedObject): - _validate_objects_for_key(shardings) -+ elif hasattr(shardings[0][1], "key") and "experts" in shardings[0][1].key: -+ continue - else: - _validate_sharding_for_key(shardings) - -diff --git a/megatron/core/optimizer/distrib_optimizer.py b/megatron/core/optimizer/distrib_optimizer.py -index 4192b0bb7..6250f1a9b 100644 ---- a/megatron/core/optimizer/distrib_optimizer.py -+++ b/megatron/core/optimizer/distrib_optimizer.py -@@ -1725,6 +1725,12 @@ class DistributedOptimizer(MixedPrecisionOptimizer): - # The optimizer state of STEP is handled - # specifically and is read from param_groups. - continue -+ if "experts" in f'{prefix}.{state_key}.{sharded_metadata.key}': -+ try: -+ from megatron.core.parallel_state import get_expert_model_parallel_rank -+ replica_id = (*replica_id[:2], get_expert_model_parallel_rank()) -+ except AssertionError: -+ pass - replace_kwargs = dict( - key=f'{prefix}.{state_key}.{sharded_metadata.key}', - data=state_ten, diff --git a/docker/npu_patch/megatron_comm.patch b/docker/npu_patch/megatron_comm.patch deleted file mode 100644 index 7da42ea46..000000000 --- a/docker/npu_patch/megatron_comm.patch +++ /dev/null @@ -1,772 +0,0 @@ -diff --git a/megatron/core/dist_checkpointing/strategies/common.py b/megatron/core/dist_checkpointing/strategies/common.py -index 41c21d93d..ef80f72d6 100644 ---- a/megatron/core/dist_checkpointing/strategies/common.py -+++ b/megatron/core/dist_checkpointing/strategies/common.py -@@ -86,7 +86,7 @@ class TorchCommonLoadStrategy(LoadCommonStrategy): - msc = MultiStorageClientFeature.import_package() - return msc.torch.load(load_path, map_location='cpu') - else: -- return torch.load(load_path, map_location='cpu') -+ return torch.load(load_path, map_location='cpu', weights_only=False) - except FileNotFoundError as e: - err_msg = f'Common file {load_path} does not exist' - if MultiStorageClientFeature.is_enabled(): -diff --git a/megatron/core/dist_checkpointing/strategies/torch.py b/megatron/core/dist_checkpointing/strategies/torch.py -index 5a1ea308d..aa701237f 100644 ---- a/megatron/core/dist_checkpointing/strategies/torch.py -+++ b/megatron/core/dist_checkpointing/strategies/torch.py -@@ -597,10 +597,12 @@ class MCoreLoadPlanner(DefaultLoadPlanner): - def _validate_global_shapes(self, metadata, sharded_tensors): - for sh_ten in sharded_tensors: - if sh_ten.key not in metadata.state_dict_metadata: -- raise KeyError( -- f"{sh_ten.key} from model not in state dict:" -- f" {sorted(metadata.state_dict_metadata.keys())}" -- ) -+ # raise KeyError( -+ # f"{sh_ten.key} from model not in state dict:" -+ # f" {sorted(metadata.state_dict_metadata.keys())}" -+ # ) -+ print(f"{sh_ten.key} from model not in state dict, will skip") -+ continue - loaded_shape = metadata.state_dict_metadata[sh_ten.key].size - expected_shape = self._expected_shape(sh_ten) - if loaded_shape != expected_shape: -@@ -630,7 +632,7 @@ class MCoreLoadPlanner(DefaultLoadPlanner): - tensor_metadata = self.metadata.state_dict_metadata - metadata_with_sizes = [ - (tensor_metadata[key], tensor_metadata[key].size, sharded_tensor) -- for key, sharded_tensor in self.allow_shape_mismatch_sharded_tensors.items() -+ for key, sharded_tensor in self.allow_shape_mismatch_sharded_tensors.items() if key in tensor_metadata - ] - try: - # Temporarily set sizes to expected shapes -@@ -959,6 +961,7 @@ class TorchDistLoadShardedStrategy(LoadShardedStrategy): - planner=MCoreLoadPlanner( - shapes_validation_sharded_tensors=flexible_shape_sharded_tensors, - allow_shape_mismatch_sharded_tensors=allow_shape_mismatch_sharded_tensors, -+ allow_partial_load=True, - ), - ) - -diff --git a/megatron/core/extensions/transformer_engine.py b/megatron/core/extensions/transformer_engine.py -index acb93ef78..d239db4ab 100644 ---- a/megatron/core/extensions/transformer_engine.py -+++ b/megatron/core/extensions/transformer_engine.py -@@ -408,6 +408,7 @@ class TELinear(te.pytorch.Linear): - ) - - for param in self.parameters(): -+ setattr(param, "parallel_mode", parallel_mode) - if is_expert: - # Reduce the gradient on the expert_data_parallel group for expert linear layers - setattr(param, "allreduce", not self.expert_parallel) -@@ -1161,6 +1162,61 @@ class TEDotProductAttention(te.pytorch.DotProductAttention): - - - if HAVE_TE and is_te_min_version("1.9.0.dev0"): -+ def ceil_div(x: int, y: int) -> int: -+ return (x + y - 1) // y -+ -+ class _FakeInt4QuantizationSTE(torch.autograd.Function): -+ @staticmethod -+ def forward(ctx, x, group_size): -+ m, n = x.shape -+ block_size_m, block_size_n = 1, group_size -+ -+ -+ m_padded = ceil_div(m, block_size_m) * block_size_m -+ n_padded = ceil_div(n, block_size_n) * block_size_n -+ -+ x_padded = torch.zeros( -+ (m_padded, n_padded), -+ dtype=x.dtype, device=x.device -+ ) -+ x_padded[:m, :n] = x -+ -+ x_view = x_padded.view( -+ m_padded // block_size_m, -+ block_size_m, -+ n_padded // block_size_n, -+ block_size_n -+ ) -+ -+ x_max = x_view.abs().float().amax(dim=(1, 3), keepdim=True) -+ q_max = 7 -+ x_scale = x_max / q_max -+ -+ x_scale = x_scale.clamp(min=1e-5) -+ -+ x_div = x_view / x_scale -+ x_round = torch.round(x_div) -+ -+ x_q_clamped = x_round.clamp(-q_max, q_max) -+ -+ x_dequant_view = x_q_clamped * x_scale -+ -+ x_dequant_full = x_dequant_view.view_as(x_padded) -+ x_out = x_dequant_full[:m, :n].contiguous().to(x.dtype) -+ -+ return x_out -+ -+ @staticmethod -+ def backward(ctx, grad_output): -+ return grad_output, None -+ -+ def fake_int4_quantization_ste(x, group_size): -+ x_out = _FakeInt4QuantizationSTE.apply(x, group_size) -+ -+ if hasattr(x, 'main_grad'): -+ x_out.main_grad = x.main_grad -+ -+ return x_out - - class TEGroupedLinear(te.pytorch.GroupedLinear): - """ -@@ -1351,6 +1407,7 @@ if HAVE_TE and is_te_min_version("1.9.0.dev0"): - _is_first_microbatch = ( - None if self.disable_parameter_transpose_cache else self.is_first_microbatch - ) -+ - out = super().forward(x, m_splits, is_first_microbatch=_is_first_microbatch) - self.is_first_microbatch = False - -@@ -1361,6 +1418,20 @@ if HAVE_TE and is_te_min_version("1.9.0.dev0"): - return out - return out, None - -+ def _get_weight_tensors(self): -+ """Get the weight tensors of the module.""" -+ weight_tensors = super()._get_weight_tensors() -+ -+ if os.getenv("OPEN_TRAINING_INT4_FAKE_QAT_FLAG", "0") == "1": -+ group_size = int(os.getenv("OPEN_TRAINING_INT4_GROUP_SIZE", "128")) -+ -+ weight_tensors = [ -+ fake_int4_quantization_ste(w, group_size) -+ for w in weight_tensors -+ ] -+ -+ return weight_tensors -+ - def _encode_extra_state(self, state): - # TE 2.0 changed the format of extra_state to be a byte tensor - if is_te_min_version("2.0.0"): -diff --git a/megatron/core/fusions/fused_mla_yarn_rope_apply.py b/megatron/core/fusions/fused_mla_yarn_rope_apply.py -index 1fd5dcfae..c9aeef1f0 100644 ---- a/megatron/core/fusions/fused_mla_yarn_rope_apply.py -+++ b/megatron/core/fusions/fused_mla_yarn_rope_apply.py -@@ -385,6 +385,7 @@ def rotary_fwd_kv_kernel( - SIN, - emb_dim: tl.constexpr, - k_dim: tl.constexpr, -+ k_dim_ceil: tl.constexpr, - v_dim: tl.constexpr, - head_num: tl.constexpr, - batch_size, -@@ -434,21 +435,27 @@ def rotary_fwd_kv_kernel( - cos_right = tl.load(COS + token_idx * emb_dim + emb_dim // 2 + tl.arange(0, emb_dim // 2)) - sin_right = tl.load(SIN + token_idx * emb_dim + emb_dim // 2 + tl.arange(0, emb_dim // 2)) - -- KV_ptr = KV + pid_m * stride_kv_seq + pid_head * BLOCK_H * stride_kv_nheads -- kv_off = tl.arange(0, BLOCK_H)[:, None] * stride_kv_nheads -- mask = kv_off < head_num * stride_kv_nheads -- k_in_off = kv_off + tl.arange(0, k_dim)[None, :] -- v_in_off = kv_off + k_dim + tl.arange(0, v_dim)[None, :] -- k = tl.load(KV_ptr + k_in_off, mask=mask) -- v = tl.load(KV_ptr + v_in_off, mask=mask) -+ KV_ptr = KV + pid_m * stride_kv_seq # + pid_head * BLOCK_H * stride_kv_nheads -+ ki_range = tl.arange(0, BLOCK_H)[:, None] + pid_head * BLOCK_H -+ kj_range = tl.arange(0, k_dim_ceil)[None, :] -+ mask_k = (ki_range < head_num) & (kj_range < k_dim) -+ mask_v = ki_range < head_num -+ k_off = ki_range * stride_kv_nheads + kj_range -+ if v_dim > 0: -+ v_off = ki_range * stride_kv_nheads + k_dim + tl.arange(0, v_dim)[None, :] -+ v = tl.load(KV_ptr + v_off, mask=mask_v) -+ else: -+ v = tl.zeros((BLOCK_H, 1), dtype=KV.dtype.element_ty) -+ k = tl.load(KV_ptr + k_off, mask=mask_k) - -- K_ptr = O_KEY + pid_m * stride_k_seq + pid_head * BLOCK_H * stride_k_nheads -- V_ptr = O_VALUE + pid_m * stride_v_seq + pid_head * BLOCK_H * stride_v_nheads -+ K_ptr = O_KEY + pid_m * stride_k_seq # + pid_head * BLOCK_H * stride_k_nheads -+ V_ptr = O_VALUE + pid_m * stride_v_seq # + pid_head * BLOCK_H * stride_v_nheads - -- k_out_off = tl.arange(0, BLOCK_H)[:, None] * stride_k_nheads + tl.arange(0, k_dim)[None, :] -- v_out_off = tl.arange(0, BLOCK_H)[:, None] * stride_v_nheads + tl.arange(0, v_dim)[None, :] -- tl.store(K_ptr + k_out_off, k, mask=mask) -- tl.store(V_ptr + v_out_off, v, mask=mask) -+ k_out_off = ki_range * stride_k_nheads + kj_range -+ tl.store(K_ptr + k_out_off, k, mask=mask_k) -+ if v_dim > 0: -+ v_out_off = ki_range * stride_v_nheads + tl.arange(0, v_dim)[None, :] -+ tl.store(V_ptr + v_out_off, v, mask=mask_v) - - EMB = K_POS_EMB + pid_m * stride_emb_seq - # x1 = t[..., 0::2], x2 = t[..., 1::2] -@@ -460,14 +467,16 @@ def rotary_fwd_kv_kernel( - x_left = x_left.expand_dims(0).broadcast_to(BLOCK_H, emb_dim // 2) - x_right = x_right.expand_dims(0).broadcast_to(BLOCK_H, emb_dim // 2) - -+ x_range = tl.arange(0, BLOCK_H)[:, None] + pid_head * BLOCK_H -+ mask_x = x_range < head_num - x_left_off = ( -- tl.arange(0, BLOCK_H)[:, None] * stride_k_nheads -+ x_range * stride_k_nheads - + k_dim - + tl.arange(0, emb_dim // 2)[None, :] - ) - x_right_off = x_left_off + emb_dim // 2 -- tl.store(K_ptr + x_left_off, x_left, mask=mask) -- tl.store(K_ptr + x_right_off, x_right, mask=mask) -+ tl.store(K_ptr + x_left_off, x_left, mask=mask_x) -+ tl.store(K_ptr + x_right_off, x_right, mask=mask_x) - - - @triton.autotune( -@@ -493,6 +502,7 @@ def rotary_bwd_kv_kernel( - SIN, - emb_dim: tl.constexpr, - k_dim: tl.constexpr, -+ k_dim_ceil: tl.constexpr, - v_dim: tl.constexpr, - head_num: tl.constexpr, - batch_size, -@@ -533,27 +543,32 @@ def rotary_bwd_kv_kernel( - else: - token_idx = _get_thd_token_idx(cu_seqlens_kv, pid_m, seq_num, cp_rank, cp_size) - -- dKV_ptr = dKV + pid_m * stride_dkv_seq + pid_head * BLOCK_H * stride_dkv_nheads -- dkv_off = tl.arange(0, BLOCK_H)[:, None] * stride_dkv_nheads -- mask = dkv_off < head_num * stride_dkv_nheads -- dk_out_off = dkv_off + tl.arange(0, k_dim)[None, :] -- dv_out_off = dkv_off + k_dim + tl.arange(0, v_dim)[None, :] -- -- dK_ptr = dK + pid_m * stride_dk_seq + pid_head * BLOCK_H * stride_dk_nheads -- dV_ptr = dV + pid_m * stride_dv_seq + pid_head * BLOCK_H * stride_dv_nheads -- dk_in_off = tl.arange(0, BLOCK_H)[:, None] * stride_dk_nheads + tl.arange(0, k_dim)[None, :] -- dv_in_off = tl.arange(0, BLOCK_H)[:, None] * stride_dv_nheads + tl.arange(0, v_dim)[None, :] -- dk = tl.load(dK_ptr + dk_in_off, mask=mask) -- dv = tl.load(dV_ptr + dv_in_off, mask=mask) -- tl.store(dKV_ptr + dk_out_off, dk, mask=mask) -- tl.store(dKV_ptr + dv_out_off, dv, mask=mask) -+ dKV_ptr = dKV + pid_m * stride_dkv_seq # + pid_head * BLOCK_H * stride_dkv_nheads -+ ki_range = tl.arange(0, BLOCK_H)[:, None] + pid_head * BLOCK_H -+ kj_range = tl.arange(0, k_dim_ceil)[None, :] -+ mask_k = (ki_range < head_num) & (kj_range < k_dim) -+ mask_v = ki_range < head_num -+ dk_out_off = ki_range * stride_dkv_nheads + kj_range -+ -+ dK_ptr = dK + pid_m * stride_dk_seq # + pid_head * BLOCK_H * stride_dk_nheads -+ dV_ptr = dV + pid_m * stride_dv_seq # + pid_head * BLOCK_H * stride_dv_nheads -+ dk_in_off = ki_range * stride_dk_nheads + kj_range -+ -+ dk = tl.load(dK_ptr + dk_in_off, mask=mask_k) -+ tl.store(dKV_ptr + dk_out_off, dk, mask=mask_k) -+ -+ if v_dim > 0: -+ dv_out_off = ki_range * stride_dkv_nheads + k_dim + tl.arange(0, v_dim)[None, :] -+ dv_in_off = ki_range * stride_dv_nheads + tl.arange(0, v_dim)[None, :] -+ dv = tl.load(dV_ptr + dv_in_off, mask=mask_v) -+ tl.store(dKV_ptr + dv_out_off, dv, mask=mask_v) - - if pid_head == 0: - x_left_accum = tl.zeros((BLOCK_H, emb_dim // 2), dtype=tl.float32) - x_right_accum = tl.zeros((BLOCK_H, emb_dim // 2), dtype=tl.float32) - for i in tl.static_range(triton.cdiv(head_num, BLOCK_H)): -- dK_ptr = dK + pid_m * stride_dk_seq + i * BLOCK_H * stride_dk_nheads -- x_off = tl.arange(0, BLOCK_H)[:, None] * stride_dk_nheads + k_dim -+ dK_ptr = dK + pid_m * stride_dk_seq # + i * BLOCK_H * stride_dk_nheads -+ x_off = tl.arange(0, BLOCK_H)[:, None] * stride_dk_nheads + k_dim + i * BLOCK_H * stride_dk_nheads - mask = x_off < head_num * stride_dk_nheads - x_left_off = x_off + tl.arange(0, emb_dim // 2)[None, :] - x_right_off = x_left_off + emb_dim // 2 -@@ -632,6 +647,7 @@ class ApplyMLARotaryEmbKV(torch.autograd.Function): - - o_key = kv.new_empty(total_seqlen, nheads, emb_dim + k_dim) - o_value = kv.new_empty(total_seqlen, nheads, v_dim) -+ k_dim_ceil = triton.next_power_of_2(k_dim) - - grid = lambda META: (total_seqlen, triton.cdiv(nheads, META["BLOCK_H"])) - rotary_fwd_kv_kernel[grid]( -@@ -643,6 +659,7 @@ class ApplyMLARotaryEmbKV(torch.autograd.Function): - sin, - emb_dim, - k_dim, -+ k_dim_ceil, - v_dim, - nheads, - batch_size, -@@ -700,6 +717,7 @@ class ApplyMLARotaryEmbKV(torch.autograd.Function): - - d_kv = dk.new_empty(total_seqlen, nheads, ctx.k_dim + ctx.v_dim) - d_emb = dk.new_empty(total_seqlen, 1, ctx.emb_dim) -+ k_dim_ceil = triton.next_power_of_2(ctx.k_dim) - - grid = lambda META: (total_seqlen, triton.cdiv(nheads, META["BLOCK_H"])) - rotary_bwd_kv_kernel[grid]( -@@ -711,6 +729,7 @@ class ApplyMLARotaryEmbKV(torch.autograd.Function): - sin, - ctx.emb_dim, - ctx.k_dim, -+ k_dim_ceil, - ctx.v_dim, - nheads, - batch_size, -diff --git a/megatron/core/models/common/language_module/language_module.py b/megatron/core/models/common/language_module/language_module.py -index 13d74aa52..060898a7a 100644 ---- a/megatron/core/models/common/language_module/language_module.py -+++ b/megatron/core/models/common/language_module/language_module.py -@@ -184,7 +184,15 @@ class LanguageModule(MegatronModule): - assert ( - column_parallel_linear is not None - ), "column_parallel_linear cannot be None when not using fused linear cross entropy." -- logits, _ = column_parallel_linear(hidden, **col_linear_kwargs) -+ # output -+ output_layer_params = {k: v.detach() for k, v in column_parallel_linear.named_parameters()} -+ output_layer_buffers = dict(column_parallel_linear.named_buffers()) -+ logits, _ = torch.func.functional_call( -+ column_parallel_linear, -+ {**output_layer_params, **output_layer_buffers}, -+ (hidden,), -+ col_linear_kwargs, -+ ) - - return self.compute_language_model_loss(labels, logits) - -diff --git a/megatron/core/models/gpt/gpt_layer_specs.py b/megatron/core/models/gpt/gpt_layer_specs.py -index e21127b87..712793853 100755 ---- a/megatron/core/models/gpt/gpt_layer_specs.py -+++ b/megatron/core/models/gpt/gpt_layer_specs.py -@@ -188,6 +188,8 @@ def get_gpt_layer_with_transformer_engine_spec( - use_kitchen: bool = False, - use_te_activation_func: bool = False, - fallback_to_eager_attn: bool = False, -+ post_self_attn_layernorm: bool = False, -+ post_mlp_layernorm: bool = False, - ) -> ModuleSpec: - """Use this spec to use lower-level Transformer Engine modules (required for fp8 training). - -@@ -260,6 +262,8 @@ def get_gpt_layer_with_transformer_engine_spec( - mlp=mlp, - sharded_state_dict_keys_map=sharded_state_dict_keys_map, - normalization=normalization, -+ post_self_attn_layernorm=post_self_attn_layernorm, -+ post_mlp_layernorm=post_mlp_layernorm, - ) - - -@@ -349,6 +353,8 @@ def get_transformer_layer_spec_for_backend( - mlp: ModuleSpec, - sharded_state_dict_keys_map: Optional[dict] = None, - normalization: Optional[str] = None, -+ post_self_attn_layernorm: bool = False, -+ post_mlp_layernorm: bool = False, - ) -> ModuleSpec: - """Helper function to get module spec for TransformerLayer""" - -@@ -371,9 +377,11 @@ def get_transformer_layer_spec_for_backend( - input_layernorm=input_layernorm, - self_attention=attention, - self_attn_bda=get_bias_dropout_add, -+ post_self_attn_layernorm=TENorm if post_self_attn_layernorm else IdentityOp, - pre_mlp_layernorm=pre_mlp_layernorm, - mlp=mlp, - mlp_bda=get_bias_dropout_add, -+ post_mlp_layernorm=TENorm if post_mlp_layernorm else IdentityOp, - sharded_state_dict_keys_map=sharded_state_dict_keys_map, - ), - ) -diff --git a/megatron/core/models/gpt/gpt_model.py b/megatron/core/models/gpt/gpt_model.py -index a1230568c..1fd52f65a 100644 ---- a/megatron/core/models/gpt/gpt_model.py -+++ b/megatron/core/models/gpt/gpt_model.py -@@ -446,6 +446,7 @@ class GPTModel(LanguageModule): - *, - inference_params: Optional[BaseInferenceContext] = None, - loss_mask: Optional[Tensor] = None, -+ mtp_kwargs: Optional[dict] = {}, - ) -> Tensor: - """Forward function of the GPT Model This function passes the input tensors - through the embedding layer, and then the decoder and finally into the post -@@ -508,6 +509,7 @@ class GPTModel(LanguageModule): - runtime_gather_output=runtime_gather_output, - extra_block_kwargs=extra_block_kwargs, - inference_context=inference_context, -+ mtp_kwargs=mtp_kwargs, - ) - - def _postprocess( -@@ -529,6 +531,7 @@ class GPTModel(LanguageModule): - runtime_gather_output=None, - extra_block_kwargs=None, - inference_context=None, -+ mtp_kwargs={}, - ): - """Postprocesses decoder hidden states to generate logits or compute loss. - -@@ -543,7 +546,8 @@ class GPTModel(LanguageModule): - output_weight = None - if self.share_embeddings_and_output_weights: - output_weight = self.shared_embedding_or_output_weight() -- if mtp_in_postprocess: -+ -+ if mtp_in_postprocess and mtp_kwargs.get('mtp_labels', None) is not None: - hidden_states = self.mtp( - input_ids=input_ids, - position_ids=position_ids, -@@ -563,13 +567,18 @@ class GPTModel(LanguageModule): - return hidden_states - - # Skip when mtp_num_layers is None or 0 -- if self.config.mtp_num_layers: -- mtp_labels = labels.clone() -+ if self.config.mtp_num_layers and mtp_kwargs.get('mtp_labels', None) is not None: -+ mtp_labels = mtp_kwargs['mtp_labels'].clone() -+ mtp_labels, _ = roll_tensor(mtp_labels, shifts=-1, dims=-1, cp_group=self.cp_group, packed_seq_params=packed_seq_params) -+ - hidden_states_list = torch.chunk(hidden_states, 1 + self.config.mtp_num_layers, dim=0) - hidden_states = hidden_states_list[0] - if loss_mask is None: - # if loss_mask is not provided, use all ones as loss_mask - loss_mask = torch.ones_like(mtp_labels) -+ else: -+ # Otherwise, roll the loss_mask to keep up with the mtp_labels -+ loss_mask, _ = roll_tensor(loss_mask, shifts=-1, dims=-1, cp_group=self.cp_group, packed_seq_params=packed_seq_params) - for mtp_layer_number in range(self.config.mtp_num_layers): - # Calc loss for the current Multi-Token Prediction (MTP) layers. - mtp_labels, _ = roll_tensor( -@@ -595,7 +604,7 @@ class GPTModel(LanguageModule): - sequence_parallel_enabled=self.output_layer.sequence_parallel, - column_parallel_linear=self.output_layer, - col_linear_kwargs={ -- 'weight': output_weight, -+ 'weight': output_weight.detach() if output_weight else None, - 'runtime_gather_output': runtime_gather_output, - }, - ) -diff --git a/megatron/core/optimizer/distrib_optimizer.py b/megatron/core/optimizer/distrib_optimizer.py -index 6e093f96f..eac21a3ea 100644 ---- a/megatron/core/optimizer/distrib_optimizer.py -+++ b/megatron/core/optimizer/distrib_optimizer.py -@@ -677,6 +677,8 @@ class DistributedOptimizer(MixedPrecisionOptimizer): - # TE FusedAdam will not accumulate step for empty param groups, so we need to - # align the step across param groups. - param_group["step"] = int(step) -+ if "step" in param_group and param_group["step"] is None: -+ del param_group["step"] - - # Grad scaler state. - if self.grad_scaler: -@@ -1646,6 +1648,8 @@ class DistributedOptimizer(MixedPrecisionOptimizer): - if key == 'padding': - tensors[key] = LocalNonpersistentObject(tensors[key]) - continue -+ if key == 'step': -+ continue - assert tensors[key].shape == (gbuf_local_end - gbuf_local_start,), ( - tensors[key].shape, - gbuf_local_start, -diff --git a/megatron/core/parallel_state.py b/megatron/core/parallel_state.py -index a273002b9..4f821cfd5 100644 ---- a/megatron/core/parallel_state.py -+++ b/megatron/core/parallel_state.py -@@ -11,6 +11,7 @@ from typing import Callable, List, Optional - - import numpy as np - import torch -+import torch.distributed as dist - - from .utils import GlobalMemoryBuffer, is_torch_min_version - -diff --git a/megatron/core/pipeline_parallel/p2p_communication.py b/megatron/core/pipeline_parallel/p2p_communication.py -index ac839c21f..f18309217 100644 ---- a/megatron/core/pipeline_parallel/p2p_communication.py -+++ b/megatron/core/pipeline_parallel/p2p_communication.py -@@ -26,22 +26,22 @@ def _batched_p2p_ops( - ops = [] - if tensor_send_prev is not None: - send_prev_op = torch.distributed.P2POp( -- torch.distributed.isend, tensor_send_prev, prev_pipeline_rank, group -+ torch.distributed.isend, tensor_send_prev, prev_pipeline_rank, - ) - ops.append(send_prev_op) - if tensor_recv_prev is not None: - recv_prev_op = torch.distributed.P2POp( -- torch.distributed.irecv, tensor_recv_prev, prev_pipeline_rank, group -+ torch.distributed.irecv, tensor_recv_prev, prev_pipeline_rank, - ) - ops.append(recv_prev_op) - if tensor_send_next is not None: - send_next_op = torch.distributed.P2POp( -- torch.distributed.isend, tensor_send_next, next_pipeline_rank, group -+ torch.distributed.isend, tensor_send_next, next_pipeline_rank, - ) - ops.append(send_next_op) - if tensor_recv_next is not None: - recv_next_op = torch.distributed.P2POp( -- torch.distributed.irecv, tensor_recv_next, next_pipeline_rank, group -+ torch.distributed.irecv, tensor_recv_next, next_pipeline_rank, - ) - ops.append(recv_next_op) - if len(ops) > 0: -diff --git a/megatron/core/transformer/moe/moe_utils.py b/megatron/core/transformer/moe/moe_utils.py -index 28cff06f5..58dc4bb70 100644 ---- a/megatron/core/transformer/moe/moe_utils.py -+++ b/megatron/core/transformer/moe/moe_utils.py -@@ -587,6 +587,9 @@ def topk_routing_with_score_function( - else: - return torch.topk(scores, k=topk, dim=1) - -+ from vime.utils.routing_replay import get_routing_replay_compute_topk -+ compute_topk = get_routing_replay_compute_topk(compute_topk) -+ - if score_function == "softmax": - if use_pre_softmax: - scores = torch.softmax(logits, dim=-1, dtype=torch.float32).type_as(logits) -diff --git a/megatron/core/transformer/moe/router.py b/megatron/core/transformer/moe/router.py -index 16fc9d9af..517944f25 100644 ---- a/megatron/core/transformer/moe/router.py -+++ b/megatron/core/transformer/moe/router.py -@@ -201,6 +201,9 @@ class TopKRouter(Router): - self.global_tokens_per_expert = None - self.ga_steps = None - -+ from vime.utils.routing_replay import register_routing_replay -+ register_routing_replay(self) -+ - def _maintain_float32_expert_bias(self): - """ - Maintain the expert bias in float32. -diff --git a/megatron/core/transformer/multi_token_prediction.py b/megatron/core/transformer/multi_token_prediction.py -index a8f4abfcd..f33f6f05e 100755 ---- a/megatron/core/transformer/multi_token_prediction.py -+++ b/megatron/core/transformer/multi_token_prediction.py -@@ -6,6 +6,7 @@ from typing import Callable, List, Optional, Union - - import torch - from torch import Tensor -+import warnings - - from megatron.core import InferenceParams, parallel_state, tensor_parallel - from megatron.core.dist_checkpointing.mapping import ShardedStateDict -@@ -714,17 +715,19 @@ class MultiTokenPredictionLayer(MegatronModule): - cp_group=self.cp_group, - packed_seq_params=packed_seq_params, - ) -- position_ids, _ = roll_tensor( -- position_ids, -- shifts=-1, -- dims=-1, -- cp_group=self.cp_group, -- packed_seq_params=packed_seq_params, -- ) -+ if position_ids is not None: -+ position_ids, _ = roll_tensor( -+ position_ids, -+ shifts=-1, -+ dims=-1, -+ cp_group=self.cp_group, -+ packed_seq_params=packed_seq_params, -+ ) - # embedding - decoder_input = embedding(input_ids=input_ids, position_ids=position_ids) -+ decoder_input = decoder_input.detach() - -- hidden_states = make_viewless_tensor(inp=hidden_states, requires_grad=True, keep_graph=True) -+ hidden_states = make_viewless_tensor(inp=hidden_states, requires_grad=True, keep_graph=False) - - return input_ids, position_ids, decoder_input, hidden_states - -@@ -826,6 +829,51 @@ class MultiTokenPredictionLayer(MegatronModule): - return hidden_states - - def _checkpointed_forward(self, forward_func, *args, **kwargs): -+ """Wrap `forward_func` with activation checkpointing while only passing tensors. -+ -+ Non-tensor arguments (e.g., configuration objects, None) are captured via closure so -+ that checkpoint implementations never receive them directly, avoiding save_for_backward -+ issues with non-tensor inputs. -+ """ -+ -+ # TODO(jiajun): Is there any better implementation here? -+ positional_specs = [] -+ kw_specs = [] -+ tensor_args: List[torch.Tensor] = [] -+ -+ for arg in args: -+ if torch.is_tensor(arg): -+ positional_specs.append(('tensor', len(tensor_args))) -+ tensor_args.append(arg) -+ else: -+ positional_specs.append(('const', arg)) -+ -+ for key, value in kwargs.items(): -+ if torch.is_tensor(value): -+ kw_specs.append((key, ('tensor', len(tensor_args)))) -+ tensor_args.append(value) -+ else: -+ kw_specs.append((key, ('const', value))) -+ -+ def run(*flat_tensor_args): -+ rebuilt_args = [] -+ for spec_type, payload in positional_specs: -+ if spec_type == 'tensor': -+ rebuilt_args.append(flat_tensor_args[payload]) -+ else: -+ rebuilt_args.append(payload) -+ -+ rebuilt_kwargs = {} -+ for key, (spec_type, payload) in kw_specs: -+ if spec_type == 'tensor': -+ rebuilt_kwargs[key] = flat_tensor_args[payload] -+ else: -+ rebuilt_kwargs[key] = payload -+ -+ return forward_func(*rebuilt_args, **rebuilt_kwargs) -+ -+ tensor_args_tuple = tuple(tensor_args) -+ - def checkpoint_handler(): - """Determines whether to use the `te_checkpoint` or `tensor_parallel.checkpoint`""" - if self.config.fp8: -@@ -836,12 +884,11 @@ class MultiTokenPredictionLayer(MegatronModule): - self.config.distribute_saved_activations, - tensor_parallel.random.get_cuda_rng_tracker, - parallel_state.get_tensor_model_parallel_group(), -- *args, -- **kwargs, -+ *tensor_args_tuple, - ) - else: - return tensor_parallel.checkpoint( -- forward_func, self.config.distribute_saved_activations, *args, *kwargs.values() -+ run, self.config.distribute_saved_activations, *tensor_args_tuple - ) - - if self.config.recompute_method == 'uniform': -diff --git a/megatron/core/transformer/transformer_config.py b/megatron/core/transformer/transformer_config.py -index e2705bd9f..a0aa109b5 100644 ---- a/megatron/core/transformer/transformer_config.py -+++ b/megatron/core/transformer/transformer_config.py -@@ -210,6 +210,9 @@ class TransformerConfig(ModelParallelConfig): - attention_output_gate: bool = False - """Whether to apply output gate to the attention layers.""" - -+ post_self_attn_layernorm: bool = False -+ post_mlp_layernorm: bool = False -+ - test_mode: bool = False - """Whether to run real-time tests.""" - -diff --git a/megatron/core/transformer/transformer_layer.py b/megatron/core/transformer/transformer_layer.py -index 3ea405770..5a42001b9 100644 ---- a/megatron/core/transformer/transformer_layer.py -+++ b/megatron/core/transformer/transformer_layer.py -@@ -223,6 +223,7 @@ class TransformerLayerSubmodules: - input_layernorm: Union[ModuleSpec, type] = IdentityOp - self_attention: Union[ModuleSpec, type] = IdentityOp - self_attn_bda: Union[ModuleSpec, type] = IdentityFuncOp -+ post_self_attn_layernorm: Union[ModuleSpec, type] = IdentityOp - - pre_cross_attn_layernorm: Union[ModuleSpec, type] = IdentityOp - cross_attention: Union[ModuleSpec, type] = IdentityOp -@@ -231,6 +232,7 @@ class TransformerLayerSubmodules: - pre_mlp_layernorm: Union[ModuleSpec, type] = IdentityOp - mlp: Union[ModuleSpec, type] = IdentityOp - mlp_bda: Union[ModuleSpec, type] = IdentityFuncOp -+ post_mlp_layernorm: Union[ModuleSpec, type] = IdentityOp - - # Mapping for sharded tensor keys to be applied in `sharded_state_dict` method - sharded_state_dict_keys_map: Dict[str, str] = field(default_factory=dict) -@@ -310,6 +312,13 @@ class TransformerLayer(GraphableMegatronModule, BaseTransformerLayer): - # [Module 3: BiasDropoutFusion] - self.self_attn_bda = build_module(submodules.self_attn_bda) - -+ self.post_self_attn_layernorm = build_module( -+ submodules.post_self_attn_layernorm, -+ config=self.config, -+ hidden_size=self.config.hidden_size, -+ eps=self.config.layernorm_epsilon, -+ ) -+ - # [Module 4: Post SelfAttention] Optional Layernorm after self-attn - self.pre_cross_attn_layernorm = build_module( - submodules.pre_cross_attn_layernorm, -@@ -375,6 +384,13 @@ class TransformerLayer(GraphableMegatronModule, BaseTransformerLayer): - - self.is_moe_layer = isinstance(self.mlp, MoELayer) - -+ self.post_mlp_layernorm = build_module( -+ submodules.post_mlp_layernorm, -+ config=self.config, -+ hidden_size=self.config.hidden_size, -+ eps=self.config.layernorm_epsilon -+ ) -+ - self.recompute_input_layernorm = False - self.recompute_pre_mlp_layernorm = False - self.recompute_mlp = False -@@ -551,6 +567,10 @@ class TransformerLayer(GraphableMegatronModule, BaseTransformerLayer): - attention_output_with_bias[0] - ) - -+ attention_output, attention_output_bias = attention_output_with_bias -+ attention_output = self.post_self_attn_layernorm(attention_output) -+ attention_output_with_bias = (attention_output, attention_output_bias) -+ - # TODO: could we move `bias_dropout_add_exec_handler` itself - # inside the module provided in the `bias_dropout_add_spec` module? - nvtx_range_push(suffix="self_attn_bda") -@@ -677,6 +697,10 @@ class TransformerLayer(GraphableMegatronModule, BaseTransformerLayer): - else: - mlp_output_with_bias = self.mlp(pre_mlp_layernorm_output) - -+ mlp_output, mlp_output_bias = mlp_output_with_bias -+ mlp_output = self.post_mlp_layernorm(mlp_output) -+ mlp_output_with_bias = (mlp_output, mlp_output_bias) -+ - if self.recompute_pre_mlp_layernorm: - # discard the output of the pre-mlp layernorm and register the recompute - # as a gradient hook of mlp_output_with_bias[0] -diff --git a/megatron/training/arguments.py b/megatron/training/arguments.py -index b267c8a81..83736acdc 100644 ---- a/megatron/training/arguments.py -+++ b/megatron/training/arguments.py -@@ -1398,6 +1398,9 @@ def core_transformer_config_from_args(args, config_class=None): - - kw_args['inference_sampling_seed'] = args.seed - -+ kw_args['post_self_attn_layernorm'] = args.post_self_attn_layernorm -+ kw_args['post_mlp_layernorm'] = args.post_mlp_layernorm -+ - # handle quantization config - # NOTE: Kitchen arguments are only added to the namespace when - # Kitchen library is available. -@@ -1764,6 +1767,12 @@ def _add_network_size_args(parser): - action='store_true', - help='If set, use original BERT residula connection ' - 'ordering.') -+ group.add_argument('--post-self-attn-layernorm', action='store_true', -+ help='If set, use post self attention layernorm.') -+ group.add_argument('--post-mlp-layernorm', action='store_true', -+ help='If set, use post MLP layernorm.') -+ group.add_argument('--use-gated-attention', action='store_true', -+ help='If set, use gated attention as in Qwen3Next') - group.add_argument('--openai-gelu', action='store_true', - help='Use OpenAIs GeLU implementation. This option' - 'should not be used unless for backward compatibility' -diff --git a/megatron/training/tokenizer/tokenizer.py b/megatron/training/tokenizer/tokenizer.py -index 13b7526ca..6c590f653 100644 ---- a/megatron/training/tokenizer/tokenizer.py -+++ b/megatron/training/tokenizer/tokenizer.py -@@ -136,7 +136,7 @@ class _HuggingFaceTokenizer(MegatronLegacyTokenizer): - # TODO(bnorick): download tokenizer once to lustre and use force offline to make sure all tasks read it from there - self._tokenizer = transformers.AutoTokenizer.from_pretrained( - pretrained_model_name_or_path=pretrained_model_name_or_path, -- trust_remote_code=trust_remote_code, -+ trust_remote_code=True, - **kwargs, - ) - self._vocab = self._tokenizer.get_vocab() diff --git a/docker/npu_patch/mindspeed.patch b/docker/npu_patch/mindspeed.patch deleted file mode 100644 index 0585af488..000000000 --- a/docker/npu_patch/mindspeed.patch +++ /dev/null @@ -1,280 +0,0 @@ -diff --git a/mindspeed/core/fusions/fused_moe_permute.py b/mindspeed/core/fusions/fused_moe_permute.py -index bb007b44..98708a5b 100644 ---- a/mindspeed/core/fusions/fused_moe_permute.py -+++ b/mindspeed/core/fusions/fused_moe_permute.py -@@ -100,9 +100,9 @@ def sort_chunks_by_idxs_wrapper(fn): - def moe_alltoall_token_dispatcher_init_wrapper(fn): - @wraps(fn) - def wrapper( -- self, num_local_experts, local_expert_indices, config, model_comm_pgs=None -+ self, num_local_experts, local_expert_indices, config, pg_collection=None - ) -> None: -- fn(self, num_local_experts, local_expert_indices, config, model_comm_pgs) -+ fn(self, num_local_experts, local_expert_indices, config, pg_collection) - # Since fused_sort_chunks_by_index is not currently supported, set self.permute_idx_device to None - self.permute_idx_device = None - input_chunk_idxs = torch.arange( -diff --git a/mindspeed/core/megatron_basic/arguments_basic.py b/mindspeed/core/megatron_basic/arguments_basic.py -index 8ea25b9f..7853bce4 100644 ---- a/mindspeed/core/megatron_basic/arguments_basic.py -+++ b/mindspeed/core/megatron_basic/arguments_basic.py -@@ -113,3 +113,35 @@ def transformer_config_init_wrapper(fn): - fn(self, *args, **known_config) - - return wrapper -+ -+ -+def transformer_config_getattr(self, name): -+ """Resolve MindSpeed extension fields for configs created before patching.""" -+ full_args = vars(get_full_args()) -+ if name in full_args: -+ return full_args[name] -+ raise AttributeError(f"'{type(self).__name__}' object has no attribute '{name}'") -+ -+ -+def transformer_config_init_subclass(cls, **kwargs): -+ mutable_types = (list, dict, set, bytearray) -+ unknown_config = {} -+ full_args = vars(get_full_args()).copy() -+ full_args.update(kwargs) -+ -+ config_key = inspect.signature(cls).parameters -+ for key, value in full_args.items(): -+ if key not in config_key: -+ unknown_config[key] = value -+ -+ for key, value in unknown_config.items(): -+ if not hasattr(cls, key): -+ cls.__annotations__[key] = type(value) -+ value = field(default_factory=value) if callable(value) and not isinstance(value, type) else value -+ if callable(value) and not isinstance(value, type): -+ value = field(default_factory=value) -+ elif type(value) in mutable_types: -+ value = field(default_factory=lambda: value) -+ else: -+ value = value -+ setattr(cls, key, value) -diff --git a/mindspeed/features_manager/megatron_basic/megatron_basic.py b/mindspeed/features_manager/megatron_basic/megatron_basic.py -index 355913e1..41678a03 100644 ---- a/mindspeed/features_manager/megatron_basic/megatron_basic.py -+++ b/mindspeed/features_manager/megatron_basic/megatron_basic.py -@@ -43,8 +43,13 @@ class MegatronBasicFeature(MindSpeedFeature): - - def register_mcore_basic_patches(self, pm, args): - # configuration patches -- from mindspeed.core.megatron_basic.arguments_basic import transformer_config_init_wrapper, transformer_config_post_init_wrapper -+ from mindspeed.core.megatron_basic.arguments_basic import (transformer_config_init_wrapper, -+ transformer_config_getattr, -+ transformer_config_post_init_wrapper, -+ transformer_config_init_subclass) - pm.register_patch("megatron.core.transformer.transformer_config.TransformerConfig.__init__", transformer_config_init_wrapper) -+ pm.register_patch("megatron.core.transformer.transformer_config.TransformerConfig.__init_subclass__", classmethod(transformer_config_init_subclass)) -+ pm.register_patch("megatron.core.transformer.transformer_config.TransformerConfig.__getattr__", transformer_config_getattr, create_dummy=True) - pm.register_patch("megatron.core.transformer.transformer_config.MLATransformerConfig.__init__", transformer_config_init_wrapper) - pm.register_patch("megatron.core.transformer.transformer_config.TransformerConfig.__post_init__", transformer_config_post_init_wrapper) - -diff --git a/mindspeed/patch_utils.py b/mindspeed/patch_utils.py -index a489d58c..d9328555 100644 ---- a/mindspeed/patch_utils.py -+++ b/mindspeed/patch_utils.py -@@ -2,7 +2,7 @@ import importlib - import sys - import types - from typing import List, Dict, Union -- -+import inspect - _MEGATRON_TRAINING_AVAILABLE = None - - -@@ -93,8 +93,10 @@ class Patch: - - def remove_patch(self): - for key, value in sys.modules.copy().items(): -- if 'mindspeed' in key: -+ if 'mindspeed' in key or 'torch.classes' == key: - continue -+ if inspect.isclass(self.orig_module) and hasattr(value, self.orig_module_name.split('.')[-1]): -+ value = getattr(value, self.orig_module_name.split('.')[-1]) - if self.orig_func_name is not None and hasattr(value, self.orig_func_name) \ - and id(getattr(value, self.orig_func_name)) == id(self.final_patch_func): - setattr(value, self.orig_func_name, self.orig_func) -diff --git a/mindspeed/core/fusions/fused_rope.py b/mindspeed/core/fusions/fused_rope.py -index a6f02e07..70f7cb08 100644 ---- a/mindspeed/core/fusions/fused_rope.py -+++ b/mindspeed/core/fusions/fused_rope.py -@@ -126,5 +126,6 @@ def apply_rotary_pos_emb( - freqs, - rotary_interleaved=config.rotary_interleaved, - multi_latent_attention=config.multi_latent_attention, -- mscale=mscale -+ mscale=mscale, -+ cp_group=cp_group - ) -diff --git a/mindspeed/megatron_adaptor.py b/mindspeed/megatron_adaptor.py -index f14b231d..4615590e 100644 ---- a/mindspeed/megatron_adaptor.py -+++ b/mindspeed/megatron_adaptor.py -@@ -57,6 +57,7 @@ def delete_lock_file(): - def repatch(args): - MindSpeedFeaturesManager.remove_patches() - full_args = get_full_args() -+ args = vars(args) - for k, v in args.items(): - setattr(full_args, k, v) - MindSpeedFeaturesManager.apply_features_pre_patches(full_args) -diff --git a/mindspeed/te/pytorch/attention/dot_product_attention/dot_product_attention.py b/mindspeed/te/pytorch/attention/dot_product_attention/dot_product_attention.py -index ac4eabe5..f82a581b 100644 ---- a/mindspeed/te/pytorch/attention/dot_product_attention/dot_product_attention.py -+++ b/mindspeed/te/pytorch/attention/dot_product_attention/dot_product_attention.py -@@ -330,6 +330,8 @@ class DotProductAttention(torch.nn.Module): - inference_params: Any = None, - pad_between_seqs: Optional[bool] = None, - fp8_output: Optional[bool] = False, -+ local_cp_size=None, -+ cp_group=None, - ) -> torch.Tensor: - """ - Dot Product Attention Layer. -@@ -653,6 +655,30 @@ class MindSpeedTEDotProductAttention(DotProductAttention): - packed_seq_params: PackedSeqParams = None, - ): - """Forward.""" -+ packed_seq_kwargs = ( -+ {key: getattr(packed_seq_params, key) for key in self.kept_packed_seq_params} -+ if packed_seq_params is not None -+ else {} -+ ) -+ -+ # Honor the per-call attn_mask_type. Megatron passes AttnMaskType.no_mask for the -+ # (bidirectional) vision tower and AttnMaskType.causal for the causal LM decoder. -+ # The previous code unconditionally built a triu(2048) causal mask and forwarded -+ # config.attention_mask_type, forcing CAUSAL attention even for no_mask -> the -+ # Qwen3-VL vision tower's full attention was computed causally (wrong image embeds). -+ if attn_mask_type == AttnMaskType.no_mask: -+ # Full (bidirectional) attention: no causal mask; per-image segmentation is -+ # handled by cu_seqlens in packed_seq_params (sparse_mode=0 via 'no_mask'). -+ core_attn_out = super().forward( -+ query, -+ key, -+ value, -+ None, -+ attn_mask_type='no_mask', -+ **packed_seq_kwargs, -+ ) -+ return core_attn_out -+ - if ( - attention_mask is None and - self.attn_mask_type == AttnMaskType.causal -@@ -660,12 +686,6 @@ class MindSpeedTEDotProductAttention(DotProductAttention): - self.config.sparse_mode = 2 - attention_mask = get_attention_mask(self.config) - -- packed_seq_kwargs = ( -- {key: getattr(packed_seq_params, key) for key in self.kept_packed_seq_params} -- if packed_seq_params is not None -- else {} -- ) -- - core_attn_out = super().forward( - query, - key, -diff --git a/mindspeed/core/fusions/fused_rope.py b/mindspeed/core/fusions/fused_rope.py -index 70f7cb08..15189c2d 100644 ---- a/mindspeed/core/fusions/fused_rope.py -+++ b/mindspeed/core/fusions/fused_rope.py -@@ -65,7 +65,7 @@ def apply_rotary_pos_emb_bshd( - cos_ = (torch.cos(freqs) * _mscale).to(t.dtype) - sin_ = (torch.sin(freqs) * _mscale).to(t.dtype) - -- if getattr(args, "use_fused_rotary_pos_emb"): -+ if getattr(args, "use_fused_rotary_pos_emb", False): - mode = 1 if rotary_interleaved else 0 - t = npu_rotary_position_embedding(t.contiguous(), cos_, sin_, mode).to(t.dtype) - else: -@@ -82,7 +82,7 @@ def transformer_config_post_init_wrapper(fn): - self.apply_rope_fusion = False - fn(self) - self.apply_rope_fusion = ori_apply_rope_fusion -- if ((getattr(self, "multi_head_latent_attention") or getattr(self, "multi_latent_attention")) -+ if ((getattr(self, "multi_head_latent_attention", False) or getattr(self, "multi_latent_attention", False)) - and self.rope_type == "yarn"): - self.apply_rope_fusion = False - del ori_apply_rope_fusion -diff --git a/mindspeed/core/models/common/embeddings/rotary_pos_embedding.py b/mindspeed/core/models/common/embeddings/rotary_pos_embedding.py -index c0319906..c0741691 100644 ---- a/mindspeed/core/models/common/embeddings/rotary_pos_embedding.py -+++ b/mindspeed/core/models/common/embeddings/rotary_pos_embedding.py -@@ -79,7 +79,7 @@ def apply_rotary_pos_emb_bshd(t: Tensor, freqs: Tensor, rotary_interleaved: bool - cos_ = (torch.cos(freqs) * _mscale).to(t.dtype) - sin_ = (torch.sin(freqs) * _mscale).to(t.dtype) - -- if getattr(args, "use_fused_rotary_pos_emb"): -+ if getattr(args, "use_fused_rotary_pos_emb", False): - mode = 1 if rotary_interleaved else 0 - t = npu_rotary_position_embedding(t.contiguous(), cos_, sin_, mode).to(t.dtype) - else: -diff --git a/mindspeed/features_manager/features_manager.py b/mindspeed/features_manager/features_manager.py -index ba476adc..f5924b98 100644 ---- a/mindspeed/features_manager/features_manager.py -+++ b/mindspeed/features_manager/features_manager.py -@@ -43,7 +43,10 @@ class MindSpeedFeaturesManager: - post_validate_features_args(args=args) # args.x = old_x - """ - for feature in cls.FEATURES_LIST: -- feature.pre_validate_args(args) -+ try: -+ feature.pre_validate_args(args) -+ except AttributeError: -+ pass - - @classmethod - def post_validate_features_args(cls, args): -@@ -54,13 +57,19 @@ class MindSpeedFeaturesManager: - post_validate_features_args(args=args) # args.x = old_x - """ - for feature in cls.FEATURES_LIST: -- feature.post_validate_args(args) -+ try: -+ feature.post_validate_args(args) -+ except AttributeError: -+ pass - - @classmethod - def validate_features_args(cls, args): - """Validate arguments of all features.""" - for feature in cls.FEATURES_LIST: -- feature.validate_args(args) -+ try: -+ feature.validate_args(args) -+ except AttributeError: -+ pass - - @classmethod - def remove_patches(cls): -diff --git a/mindspeed/features_manager/moe/fb_overlap.py b/mindspeed/features_manager/moe/fb_overlap.py -index 6c76d891..e099c51d 100644 ---- a/mindspeed/features_manager/moe/fb_overlap.py -+++ b/mindspeed/features_manager/moe/fb_overlap.py -@@ -16,6 +16,8 @@ class MoEFwdBwdOverlapFeature(MindSpeedFeature): - group.add_argument('--moe-unperm2-mem-optim-swap', action='store_true') - - def validate_args(self, args): -+ if not hasattr(args, 'moe_fb_overlap'): -+ return - self.incompatible_check(args, 'moe_alltoall_overlap_comm') - self.incompatible_check(args, 'overlap_grad_reduce') - self.incompatible_check(args, 'moe_hierarchical_alltoallv') -diff --git a/mindspeed/features_manager/moe/moe_zero_memory.py b/mindspeed/features_manager/moe/moe_zero_memory.py -index 4d95e64e..94f26109 100644 ---- a/mindspeed/features_manager/moe/moe_zero_memory.py -+++ b/mindspeed/features_manager/moe/moe_zero_memory.py -@@ -23,6 +23,8 @@ class MoEZeroMemoryFeature(MindSpeedFeature): - 'in each pp stage.') - - def pre_validate_args(self, args): -+ if not hasattr(args, "moe_zero_memory_num_layers") or not hasattr(args, "moe_zero_memory"): -+ return - #Zero Memory check. - if args.moe_zero_memory_num_layers is not None: - num_layers_per_pipeline_stage = args.num_layers // args.pipeline_model_parallel_size diff --git a/docker/npu_patch/series.conf b/docker/npu_patch/series.conf index 2729cfe34..24fd3a586 100644 --- a/docker/npu_patch/series.conf +++ b/docker/npu_patch/series.conf @@ -11,7 +11,5 @@ # Ordinary patch changes must not add patch-specific logic to the CI script. /vllm-workspace/vllm|vllm.patch|docker/npu_patch/vllm.patch /vllm-workspace/vllm-ascend|vllm-ascend.patch|docker/npu_patch/vllm-ascend.patch -/root/Megatron-LM|megatron_comm.patch|docker/npu_patch/megatron_comm.patch /root/Megatron-LM|megatron.patch|docker/npu_patch/megatron.patch /root/Megatron-Bridge|megatron-bridge.patch|docker/npu_patch/megatron-bridge.patch -/root/MindSpeed|mindspeed.patch|docker/npu_patch/mindspeed.patch diff --git a/docker/patch/latest/vllm.patch b/docker/patch/latest/vllm.patch index 22548e6fc..e278cf3f0 100644 --- a/docker/patch/latest/vllm.patch +++ b/docker/patch/latest/vllm.patch @@ -27,4 +27,4 @@ diff --git a/vllm/v1/engine/core.py b/vllm/v1/engine/core.py + self.execute_dummy_batch() # 3) All-reduce operation to determine global unfinished reqs. - self.engines_running = self._has_global_unfinished_reqs( + self.engines_running = self._has_global_unfinished_reqs( \ No newline at end of file diff --git a/examples/retool/retool_qwen3_4b_rl.sh b/examples/retool/retool_qwen3_4b_rl.sh index 634c2d0a0..42d4eb2da 100644 --- a/examples/retool/retool_qwen3_4b_rl.sh +++ b/examples/retool/retool_qwen3_4b_rl.sh @@ -34,7 +34,7 @@ VIME_DIR="/root/vime" source /usr/local/Ascend/driver/bin/setenv.bash source /usr/local/Ascend/ascend-toolkit/set_env.sh source /usr/local/Ascend/nnal/atb/set_env.sh -export PYTHONPATH="${VIME_DIR}:${VIME_DIR}/examples/retool:/root/Megatron-LM:/root/vllm:/root/vllm-ascend:/root/Megatron-Bridge:/root/mbridge:/root/MindSpeed:/usr/local/Ascend/ascend-toolkit/latest/python/site-packages:/usr/local/Ascend/ascend-toolkit/latest/tools/ms_fmk_transplt/torch_npu_bridge:${PYTHONPATH}" +export PYTHONPATH="${VIME_DIR}:${VIME_DIR}/examples/retool:/root/Megatron-LM:/root/vllm:/root/vllm-ascend:/root/Megatron-Bridge:/root/mbridge:/root/MegatronAdaptor:/root/TransformerEngineNPU:/usr/local/Ascend/ascend-toolkit/latest/python/site-packages:/usr/local/Ascend/ascend-toolkit/latest/tools/ms_fmk_transplt/torch_npu_bridge:${PYTHONPATH}" export PYTHONUNBUFFERED=1 export PYTORCH_NPU_ALLOC_CONF=expandable_segments:False export CUDA_DEVICE_MAX_CONNECTIONS=1 @@ -169,7 +169,7 @@ ray start --head \ RUNTIME_ENV_JSON=$(cat << 'EOF' { "env_vars": { - "PYTHONPATH": "${VIME_DIR}:${VIME_DIR}/examples/retool:/root/Megatron-LM:/root/vllm:/root/vllm-ascend:/root/Megatron-Bridge:/root/mbridge:/root/MindSpeed:/usr/local/Ascend/ascend-toolkit/latest/python/site-packages:/usr/local/Ascend/ascend-toolkit/latest/tools/ms_fmk_transplt/torch_npu_bridge", + "PYTHONPATH": "${VIME_DIR}:${VIME_DIR}/examples/retool:/root/Megatron-LM:/root/vllm:/root/vllm-ascend:/root/Megatron-Bridge:/root/mbridge:/root/MegatronAdaptor:/root/TransformerEngineNPU:/usr/local/Ascend/ascend-toolkit/latest/python/site-packages:/usr/local/Ascend/ascend-toolkit/latest/tools/ms_fmk_transplt/torch_npu_bridge", "CUDA_DEVICE_MAX_CONNECTIONS": "1", "HCCL_HOST_SOCKET_PORT_RANGE": "60000-60050", "HCCL_NPU_SOCKET_PORT_RANGE": "61000-61050", diff --git a/examples/retool/retool_qwen3_4b_sft.sh b/examples/retool/retool_qwen3_4b_sft.sh index 7ada44cfb..cf067ab80 100644 --- a/examples/retool/retool_qwen3_4b_sft.sh +++ b/examples/retool/retool_qwen3_4b_sft.sh @@ -34,7 +34,7 @@ VIME_DIR="/root/vime" source /usr/local/Ascend/driver/bin/setenv.bash source /usr/local/Ascend/ascend-toolkit/set_env.sh source /usr/local/Ascend/nnal/atb/set_env.sh -export PYTHONPATH="${VIME_DIR}:${VIME_DIR}/examples/retool:/root/Megatron-LM:/root/vllm:/root/vllm-ascend:/root/Megatron-Bridge:/root/mbridge:/root/MindSpeed:/usr/local/Ascend/ascend-toolkit/latest/python/site-packages:/usr/local/Ascend/ascend-toolkit/latest/tools/ms_fmk_transplt/torch_npu_bridge:${PYTHONPATH}" +export PYTHONPATH="${VIME_DIR}:${VIME_DIR}/examples/retool:/root/Megatron-LM:/root/vllm:/root/vllm-ascend:/root/Megatron-Bridge:/root/mbridge:/root/MegatronAdaptor:/root/TransformerEngineNPU:/usr/local/Ascend/ascend-toolkit/latest/python/site-packages:/usr/local/Ascend/ascend-toolkit/latest/tools/ms_fmk_transplt/torch_npu_bridge:${PYTHONPATH}" export PYTHONUNBUFFERED=1 export PYTORCH_NPU_ALLOC_CONF=expandable_segments:False export CUDA_DEVICE_MAX_CONNECTIONS=1 @@ -138,7 +138,7 @@ ray start --head \ RUNTIME_ENV_JSON=$(cat << 'EOF' { "env_vars": { - "PYTHONPATH": "${VIME_DIR}:${VIME_DIR}/examples/retool:/root/Megatron-LM:/root/vllm:/root/vllm-ascend:/root/Megatron-Bridge:/root/mbridge:/root/MindSpeed:/usr/local/Ascend/ascend-toolkit/latest/python/site-packages:/usr/local/Ascend/ascend-toolkit/latest/tools/ms_fmk_transplt/torch_npu_bridge", + "PYTHONPATH": "${VIME_DIR}:${VIME_DIR}/examples/retool:/root/Megatron-LM:/root/vllm:/root/vllm-ascend:/root/Megatron-Bridge:/root/mbridge:/root/MegatronAdaptor:/root/TransformerEngineNPU:/usr/local/Ascend/ascend-toolkit/latest/python/site-packages:/usr/local/Ascend/ascend-toolkit/latest/tools/ms_fmk_transplt/torch_npu_bridge", "CUDA_DEVICE_MAX_CONNECTIONS": "1", "HCCL_HOST_SOCKET_PORT_RANGE": "60000-60050", "HCCL_NPU_SOCKET_PORT_RANGE": "61000-61050", diff --git a/examples/search-r1/run_qwen3_4b_npu.sh b/examples/search-r1/run_qwen3_4b_npu.sh index df003516b..8f2aa6dea 100644 --- a/examples/search-r1/run_qwen3_4b_npu.sh +++ b/examples/search-r1/run_qwen3_4b_npu.sh @@ -34,7 +34,7 @@ VIME_DIR="/root/vime" source /usr/local/Ascend/driver/bin/setenv.bash source /usr/local/Ascend/ascend-toolkit/set_env.sh source /usr/local/Ascend/nnal/atb/set_env.sh -export PYTHONPATH="${VIME_DIR}/examples/search-r1:/root/Megatron-LM:/root/vllm:/root/vllm-ascend:${VIME_DIR}:/root/Megatron-Bridge:/root/mbridge:/root/MindSpeed:/usr/local/Ascend/ascend-toolkit/latest/python/site-packages:/usr/local/Ascend/ascend-toolkit/latest/tools/ms_fmk_transplt/torch_npu_bridge:${PYTHONPATH}" +export PYTHONPATH="${VIME_DIR}/examples/search-r1:/root/Megatron-LM:/root/vllm:/root/vllm-ascend:${VIME_DIR}:/root/Megatron-Bridge:/root/mbridge:/root/MegatronAdaptor:/root/TransformerEngineNPU:/usr/local/Ascend/ascend-toolkit/latest/python/site-packages:/usr/local/Ascend/ascend-toolkit/latest/tools/ms_fmk_transplt/torch_npu_bridge:${PYTHONPATH}" export PYTHONUNBUFFERED=1 export PYTORCH_NPU_ALLOC_CONF=expandable_segments:False export CUDA_DEVICE_MAX_CONNECTIONS=1 @@ -176,7 +176,7 @@ ray start --head \ RUNTIME_ENV_JSON=$(cat << 'EOF' { "env_vars": { - "PYTHONPATH": "${VIME_DIR}/examples/search-r1:/root/Megatron-LM:/root/vllm:/root/vllm-ascend:${VIME_DIR}:/root/Megatron-Bridge:/root/mbridge:/root/MindSpeed:/usr/local/Ascend/ascend-toolkit/latest/python/site-packages:/usr/local/Ascend/ascend-toolkit/latest/tools/ms_fmk_transplt/torch_npu_bridge", + "PYTHONPATH": "${VIME_DIR}/examples/search-r1:/root/Megatron-LM:/root/vllm:/root/vllm-ascend:${VIME_DIR}:/root/Megatron-Bridge:/root/mbridge:/root/MegatronAdaptor:/root/TransformerEngineNPU:/usr/local/Ascend/ascend-toolkit/latest/python/site-packages:/usr/local/Ascend/ascend-toolkit/latest/tools/ms_fmk_transplt/torch_npu_bridge", "CUDA_DEVICE_MAX_CONNECTIONS": "1", "HCCL_HOST_SOCKET_PORT_RANGE": "60000-60050", "HCCL_NPU_SOCKET_PORT_RANGE": "61000-61050", diff --git a/tests/unit/backends/megatron_utils/update_weight/test_update_weight_from_tensor.py b/tests/unit/backends/megatron_utils/update_weight/test_update_weight_from_tensor.py index bd178d9d9..143af7645 100644 --- a/tests/unit/backends/megatron_utils/update_weight/test_update_weight_from_tensor.py +++ b/tests/unit/backends/megatron_utils/update_weight/test_update_weight_from_tensor.py @@ -14,7 +14,7 @@ MODULE_PATH = "vime.backends.megatron_utils.update_weight.update_weight_from_tensor" -_PURGE_PREFIXES = ("megatron", "mindspeed", "vime.backends.megatron_utils") +_PURGE_PREFIXES = ("megatron", "megatron_adaptor", "vime.backends.megatron_utils") def _collect_subtree(prefix: str) -> list[str]: diff --git a/train.py b/train.py index 86d9c5e82..949ef1609 100644 --- a/train.py +++ b/train.py @@ -7,7 +7,7 @@ from vime.utils.misc import should_run_periodic_action if is_npu(): - import mindspeed.megatron_adaptor # noqa: F401 + import megatron_adaptor # noqa: F401 def train(args): diff --git a/train_async.py b/train_async.py index 9152fec78..93aad67b3 100644 --- a/train_async.py +++ b/train_async.py @@ -7,7 +7,7 @@ from vime.utils.misc import should_run_periodic_action if is_npu(): - import mindspeed.megatron_adaptor # noqa: F401 + import megatron_adaptor # noqa: F401 # The framework supports other asynchronous approaches such as fully async (which is shown in examples/full_async). diff --git a/vime/backends/megatron_utils/__init__.py b/vime/backends/megatron_utils/__init__.py index 2619c8cd4..1497b5a48 100644 --- a/vime/backends/megatron_utils/__init__.py +++ b/vime/backends/megatron_utils/__init__.py @@ -10,7 +10,10 @@ from vime.utils.common import is_npu if is_npu(): - import mindspeed.megatron_adaptor # noqa: F401 + # MegatronAdaptor must run before Megatron imports so its dummy NPU modules + # are bound in Megatron tensor-parallel modules. + import megatron_adaptor # noqa: F401 + from . import npu_attention_patch # noqa: F401 try: import deep_ep diff --git a/vime/backends/megatron_utils/actor.py b/vime/backends/megatron_utils/actor.py index e8781cb09..68aac0f71 100644 --- a/vime/backends/megatron_utils/actor.py +++ b/vime/backends/megatron_utils/actor.py @@ -14,8 +14,20 @@ if is_npu(): import importlib + import megatron_adaptor # noqa: F401 + importlib.import_module("vime.backends.megatron_utils.npu_attention_patch") - from mindspeed.megatron_adaptor import repatch + from megatron_adaptor.features_manager.features_manager import FeaturesManager + from megatron_adaptor.utils.args_utils import get_full_args + + def _repatch_megatron_adaptor(args): + """Reapply MegatronAdaptor features after VIME has finalized args.""" + full_args = get_full_args() + for key, value in vars(args).items(): + setattr(full_args, key, value) + FeaturesManager.remove_patches() + FeaturesManager.apply_features_pre_patches(full_args) + FeaturesManager.apply_features_patches(full_args) _orig_npu_empty_cache = torch.npu.empty_cache @@ -79,7 +91,7 @@ def init( init(args) if is_npu(): - repatch(args) + _repatch_megatron_adaptor(args) if is_megatron_main_rank(): init_tracking(args, primary=False, role=role) diff --git a/vime/backends/megatron_utils/npu_attention_patch.py b/vime/backends/megatron_utils/npu_attention_patch.py index be35fe17c..184f5e717 100644 --- a/vime/backends/megatron_utils/npu_attention_patch.py +++ b/vime/backends/megatron_utils/npu_attention_patch.py @@ -79,4 +79,7 @@ def npu_dot_product_attention_forward( from megatron.core.transformer.dot_product_attention import DotProductAttention DotProductAttention.forward = npu_dot_product_attention_forward -print("[NPU PATCH] DotProductAttention.forward replaced with npu_fusion_attention BEFORE mindspeed import", flush=True) +print( + "[NPU PATCH] DotProductAttention.forward replaced with npu_fusion_attention " "BEFORE megatron_adaptor import", + flush=True, +) diff --git a/vime/backends/megatron_utils/update_weight/update_weight_from_tensor.py b/vime/backends/megatron_utils/update_weight/update_weight_from_tensor.py index f3ec65292..c1052e4ea 100644 --- a/vime/backends/megatron_utils/update_weight/update_weight_from_tensor.py +++ b/vime/backends/megatron_utils/update_weight/update_weight_from_tensor.py @@ -269,7 +269,7 @@ class _VLLMHijack: MoE weight_loader missing on EP (a vLLM bug where w13_weight/w2_weight params lack weight_loader attr when EP is enabled). - Patches ApplyRotaryEmb.__init__ to skip flash_attn import - (mindspeed/megatron backends introduce flash_attn as a dummy module, + (Megatron/NPU backends introduce flash_attn as a dummy module, but vllm_ascend does not use it). """ diff --git a/vime/utils/arguments.py b/vime/utils/arguments.py index 7b3ae8968..968bec729 100644 --- a/vime/utils/arguments.py +++ b/vime/utils/arguments.py @@ -1798,7 +1798,10 @@ def vime_validate_args(args): if args.use_critic: args.offload_train = True - if args.offload_train: + # Megatron mainline uses torch_memory_saver regions for these buffers. + # On NPU, the actor already owns the outer training region; opening nested + # NPU mem-pool regions causes beginAllocateToPool() failures. + if args.offload_train and not is_npu(): args.disable_grad_buffers_cpu_backup = True args.disable_param_buffers_cpu_backup = True diff --git a/vime/utils/external_utils/launch.py b/vime/utils/external_utils/launch.py index f31ef2b97..a849b7476 100644 --- a/vime/utils/external_utils/launch.py +++ b/vime/utils/external_utils/launch.py @@ -79,7 +79,8 @@ def _detect_npu() -> bool: env={ "PYTHONPATH": ( "/root/Megatron-LM:/root/vime:" - "/root/Megatron-Bridge/src:/root/mbridge:/root/MindSpeed:" + "/root/Megatron-Bridge/src:/root/mbridge:" + "/root/MegatronAdaptor:/root/TransformerEngineNPU:" "/usr/local/Ascend/ascend-toolkit/latest/python/site-packages:" "/usr/local/Ascend/ascend-toolkit/latest/tools/ms_fmk_transplt/torch_npu_bridge" ), From dbbe1005c066f1f64b96761a371984c0da6584de Mon Sep 17 00:00:00 2001 From: wangxiaoxin-sherie Date: Thu, 27 Aug 2026 10:47:48 +0800 Subject: [PATCH 02/12] fix(npu): apply common Megatron patch before NPU patch Signed-off-by: wangxiaoxin-sherie --- docker/Dockerfile.npu | 13 ++++++++++++- 1 file changed, 12 insertions(+), 1 deletion(-) diff --git a/docker/Dockerfile.npu b/docker/Dockerfile.npu index c2bdead55..3a1b370d8 100644 --- a/docker/Dockerfile.npu +++ b/docker/Dockerfile.npu @@ -32,6 +32,7 @@ ENV SOC_VERSION=$SOC_VERSION \ # PATCH MAINTENANCE: keep patch COPY/apply operations and # docker/npu_patch/series.conf synchronized. COPY docker/npu_patch /opt/npu_patch +COPY docker/patch/latest/megatron.patch /opt/vime_patch/megatron.patch RUN git config --global http.sslVerify false @@ -78,8 +79,18 @@ RUN git clone https://github.com/ISEEKYAN/mbridge.git /root/mbridge && \ git -C /root/mbridge checkout "${MBRIDGE_COMMIT}" # Apply NPU training-stack patches from the build-context snapshot. -RUN git -C /root/Megatron-LM apply --whitespace=nowarn \ +# The NPU Megatron patch is based on the common Vime Megatron patch, so the +# common patch must be applied first. +RUN git -C /root/Megatron-LM apply --check --whitespace=nowarn \ + /opt/vime_patch/megatron.patch && \ + git -C /root/Megatron-LM apply --whitespace=nowarn \ + /opt/vime_patch/megatron.patch && \ + git -C /root/Megatron-LM apply --check --whitespace=nowarn \ /opt/npu_patch/megatron.patch && \ + git -C /root/Megatron-LM apply --whitespace=nowarn \ + /opt/npu_patch/megatron.patch && \ + git -C /root/Megatron-Bridge apply --check --whitespace=nowarn \ + /opt/npu_patch/megatron-bridge.patch && \ git -C /root/Megatron-Bridge apply --whitespace=nowarn \ /opt/npu_patch/megatron-bridge.patch From 9e2079d88b99b731b1bde77b93af1271d293f244 Mon Sep 17 00:00:00 2001 From: wangxiaoxin-sherie Date: Thu, 27 Aug 2026 11:05:21 +0800 Subject: [PATCH 03/12] fix(npu): complete and pin training dependencies Signed-off-by: wangxiaoxin-sherie --- docker/Dockerfile.npu | 10 +++++++--- 1 file changed, 7 insertions(+), 3 deletions(-) diff --git a/docker/Dockerfile.npu b/docker/Dockerfile.npu index 3a1b370d8..32f3d42ce 100644 --- a/docker/Dockerfile.npu +++ b/docker/Dockerfile.npu @@ -9,6 +9,8 @@ WORKDIR /root ARG MEGATRON_COMMIT=1dcf0dafa884ad52ffb243625717a3471643e087 ARG MEGATRON_BRIDGE_COMMIT=3fd3768045422d0aa5c97e90a4e6c659aea9acb9 +ARG MEGATRON_ADAPTOR_COMMIT=15582addff3f3d4680e350826fa70d012b475509 +ARG TRANSFORMER_ENGINE_NPU_COMMIT=d743c83d060d5edc48867ecb9e93ec80d81860e4 ARG MBRIDGE_COMMIT=89eb10887887bc74853f89a4de258c0702932a1c ARG SOC_VERSION="ascend910_9391" ARG PIP_INDEX_URL="https://mirrors.tuna.tsinghua.edu.cn/pypi/web/simple" @@ -73,7 +75,9 @@ RUN git clone --branch bridge https://github.com/radixark/Megatron-Bridge.git \ git -C /root/Megatron-Bridge checkout "${MEGATRON_BRIDGE_COMMIT}" RUN git clone https://gitcode.com/Ascend/MegatronAdaptor.git /root/MegatronAdaptor && \ - git clone https://gitcode.com/Ascend/TransformerEngineNPU.git /root/TransformerEngineNPU + git -C /root/MegatronAdaptor checkout "${MEGATRON_ADAPTOR_COMMIT}" && \ + git clone https://gitcode.com/Ascend/TransformerEngineNPU.git /root/TransformerEngineNPU && \ + git -C /root/TransformerEngineNPU checkout "${TRANSFORMER_ENGINE_NPU_COMMIT}" RUN git clone https://github.com/ISEEKYAN/mbridge.git /root/mbridge && \ git -C /root/mbridge checkout "${MBRIDGE_COMMIT}" @@ -96,7 +100,8 @@ RUN git -C /root/Megatron-LM apply --check --whitespace=nowarn \ # Megatron-Bridge is used directly from PYTHONPATH. Installing its package # metadata would pull CUDA-only dependencies into the Ascend environment. -RUN pip install --no-build-isolation "nvidia-modelopt[torch]>=0.37.0" && \ +RUN pip install --constraint /tmp/vime-npu-constraints.txt --no-build-isolation \ + "nvidia-modelopt==0.46.0" "nvdlfw-inspect==0.2.2" && \ pip install --no-deps --no-build-isolation -e /root/mbridge && \ pip install --no-deps --no-build-isolation -e /root/Megatron-LM && \ pip install --no-deps --no-build-isolation -e /root/TransformerEngineNPU && \ @@ -121,7 +126,6 @@ RUN pip install \ RUN git clone --depth 1 --branch 2026.6.0 \ https://github.com/sgl-project/sgl-kernel-npu.git /root/sgl-kernel-npu && \ cd /root/sgl-kernel-npu && \ - bash build.sh -a kernels && \ bash build.sh -a memory-saver && \ pip install --no-deps \ output/torch_memory_saver-*.whl && \ From 0d5e89416d7b5998a15fd7fc8f368e7315b1b375 Mon Sep 17 00:00:00 2001 From: wangxiaoxin-sherie Date: Thu, 27 Aug 2026 11:17:01 +0800 Subject: [PATCH 04/12] feat(npu): support proxied GitHub builds Signed-off-by: wangxiaoxin-sherie --- docker/Dockerfile.npu | 22 ++- docker/npu_patch/README.md | 18 +- docker/npu_patch/mindspeed.patch | 280 +++++++++++++++++++++++++++++++ docker/npu_patch/series.conf | 1 + 4 files changed, 312 insertions(+), 9 deletions(-) create mode 100644 docker/npu_patch/mindspeed.patch diff --git a/docker/Dockerfile.npu b/docker/Dockerfile.npu index 32f3d42ce..dba692d9f 100644 --- a/docker/Dockerfile.npu +++ b/docker/Dockerfile.npu @@ -9,12 +9,14 @@ WORKDIR /root ARG MEGATRON_COMMIT=1dcf0dafa884ad52ffb243625717a3471643e087 ARG MEGATRON_BRIDGE_COMMIT=3fd3768045422d0aa5c97e90a4e6c659aea9acb9 +ARG MINDSPEED_COMMIT=fc63de5c48426dd019c3b3f39e65f5bdf56e4086 ARG MEGATRON_ADAPTOR_COMMIT=15582addff3f3d4680e350826fa70d012b475509 ARG TRANSFORMER_ENGINE_NPU_COMMIT=d743c83d060d5edc48867ecb9e93ec80d81860e4 ARG MBRIDGE_COMMIT=89eb10887887bc74853f89a4de258c0702932a1c ARG SOC_VERSION="ascend910_9391" ARG PIP_INDEX_URL="https://mirrors.tuna.tsinghua.edu.cn/pypi/web/simple" ARG APTMIRROR="" +ARG GITHUB_PROXY="" ENV SOC_VERSION=$SOC_VERSION \ DEBIAN_FRONTEND=noninteractive \ @@ -36,7 +38,12 @@ ENV SOC_VERSION=$SOC_VERSION \ COPY docker/npu_patch /opt/npu_patch COPY docker/patch/latest/megatron.patch /opt/vime_patch/megatron.patch -RUN git config --global http.sslVerify false +RUN if [ -n "${GITHUB_PROXY}" ]; then \ + git config --global \ + url."${GITHUB_PROXY%/}/https://github.com/".insteadOf \ + "https://github.com/"; \ + fi && \ + git config --global http.sslVerify false # System and pip dependencies. RUN if [ -n "$APTMIRROR" ];then sed -i "s@^\(deb.*\)https\?://[a-z0-9.-]*\.ubuntu\.com@\1$APTMIRROR@g" /etc/apt/sources.list; \ @@ -74,6 +81,9 @@ RUN git clone --branch bridge https://github.com/radixark/Megatron-Bridge.git \ /root/Megatron-Bridge && \ git -C /root/Megatron-Bridge checkout "${MEGATRON_BRIDGE_COMMIT}" +RUN git clone https://gitcode.com/Ascend/MindSpeed.git /root/MindSpeed && \ + git -C /root/MindSpeed checkout "${MINDSPEED_COMMIT}" + RUN git clone https://gitcode.com/Ascend/MegatronAdaptor.git /root/MegatronAdaptor && \ git -C /root/MegatronAdaptor checkout "${MEGATRON_ADAPTOR_COMMIT}" && \ git clone https://gitcode.com/Ascend/TransformerEngineNPU.git /root/TransformerEngineNPU && \ @@ -96,7 +106,9 @@ RUN git -C /root/Megatron-LM apply --check --whitespace=nowarn \ git -C /root/Megatron-Bridge apply --check --whitespace=nowarn \ /opt/npu_patch/megatron-bridge.patch && \ git -C /root/Megatron-Bridge apply --whitespace=nowarn \ - /opt/npu_patch/megatron-bridge.patch + /opt/npu_patch/megatron-bridge.patch && \ + git -C /root/MindSpeed apply --whitespace=nowarn \ + /opt/npu_patch/mindspeed.patch # Megatron-Bridge is used directly from PYTHONPATH. Installing its package # metadata would pull CUDA-only dependencies into the Ascend environment. @@ -105,7 +117,8 @@ RUN pip install --constraint /tmp/vime-npu-constraints.txt --no-build-isolation pip install --no-deps --no-build-isolation -e /root/mbridge && \ pip install --no-deps --no-build-isolation -e /root/Megatron-LM && \ pip install --no-deps --no-build-isolation -e /root/TransformerEngineNPU && \ - pip install --no-deps --no-build-isolation -e /root/MegatronAdaptor + pip install --no-deps --no-build-isolation -e /root/MegatronAdaptor && \ + pip install --no-deps --no-build-isolation -e /root/MindSpeed # Defaults to the ascend branch for local builds. Release workflows should # pass an immutable commit SHA for reproducible images. @@ -126,6 +139,7 @@ RUN pip install \ RUN git clone --depth 1 --branch 2026.6.0 \ https://github.com/sgl-project/sgl-kernel-npu.git /root/sgl-kernel-npu && \ cd /root/sgl-kernel-npu && \ + bash build.sh -a kernels && \ bash build.sh -a memory-saver && \ pip install --no-deps \ output/torch_memory_saver-*.whl && \ @@ -134,7 +148,7 @@ RUN git clone --depth 1 --branch 2026.6.0 \ # Minimal import check. RUN source /usr/local/Ascend/ascend-toolkit/set_env.sh && \ - python3 -c 'import megatron, megatron_adaptor, transformer_engine, torch_memory_saver, vime, vllm, vllm_ascend;' + python3 -c 'import megatron, mindspeed, megatron_adaptor, transformer_engine, torch_memory_saver, vime, vllm, vllm_ascend;' WORKDIR /root/vime ENTRYPOINT [] diff --git a/docker/npu_patch/README.md b/docker/npu_patch/README.md index 19caa4a74..2fd4ce2e2 100644 --- a/docker/npu_patch/README.md +++ b/docker/npu_patch/README.md @@ -11,6 +11,7 @@ This guide provides instructions for installing Vime with NPU support, including | Megatron-LM | 1dcf0dafa884ad52ffb243625717a3471643e087 | [GitHub](https://github.com/NVIDIA/Megatron-LM) | | MegatronAdaptor | main | [GitCode](https://gitcode.com/Ascend/MegatronAdaptor) | | TransformerEngineNPU | main | [GitCode](https://gitcode.com/Ascend/TransformerEngineNPU) | +| MindSpeed | fc63de5c48426dd019c3b3f39e65f5bdf56e4086 | [GitCode](https://gitcode.com/Ascend/MindSpeed) | | HDK | 25.3.RC1 | [Ascend](https://www.hiascend.com/hardware/firmware-drivers/commercial?product=7\&model=33) | | CANN | 9.0.0 | [Ascend](https://www.hiascend.com/developer/download/community/result?module=cann\&cann=9.0.0\&product=7\&model=33) | @@ -32,7 +33,6 @@ git clone --branch ascend https://github.com/vllm-project/vime.git "${WORKSPACE} export PATCH_DIR="${WORKSPACE}/vime/docker/npu_patch" ``` - #### 1. Megatron-Bridge Used via `PYTHONPATH` (no editable install); it requires `nvidia-modelopt`. @@ -49,7 +49,6 @@ git -C "${WORKSPACE}/Megatron-Bridge" apply --whitespace=nowarn "${PATCH_DIR}/me pip install --no-build-isolation "nvidia-modelopt[torch]>=0.37.0" ``` - #### 2. Megatron-LM ```bash @@ -72,8 +71,19 @@ pip install --no-deps --no-build-isolation -e ${WORKSPACE}/TransformerEngineNPU Do not install the CUDA TransformerEngine package in the same environment. +#### 4. MegatronAdaptor and TransformerEngineNPU -#### 4. Vime +```bash +export MINDSPEED_COMMIT=fc63de5c48426dd019c3b3f39e65f5bdf56e4086 +git clone https://gitcode.com/Ascend/MindSpeed.git "${WORKSPACE}/MindSpeed" +git -C "${WORKSPACE}/MindSpeed" checkout "${MINDSPEED_COMMIT}" + +git -C "${WORKSPACE}/MindSpeed" apply --whitespace=nowarn "${PATCH_DIR}/mindspeed.patch" + +pip install --no-deps --no-build-isolation -e "${WORKSPACE}/MindSpeed" +``` + +#### 5. Vime ```bash pip install -r "${WORKSPACE}/vime/requirements.txt" @@ -96,7 +106,6 @@ pip install --no-deps output/torch_memory_saver-0.0.8-cp312-cp312-linux_aarch64. #### 5. Install vLLM and vLLM Ascend - ```bash export VLLM_COMMIT=9090368b650896bf5fc990c921df7eb4c20355a5 @@ -124,4 +133,3 @@ pip install torch-npu==2.10.0 pip install torchvision==0.25.0 pip install numpy==1.26.4 ``` - diff --git a/docker/npu_patch/mindspeed.patch b/docker/npu_patch/mindspeed.patch new file mode 100644 index 000000000..0585af488 --- /dev/null +++ b/docker/npu_patch/mindspeed.patch @@ -0,0 +1,280 @@ +diff --git a/mindspeed/core/fusions/fused_moe_permute.py b/mindspeed/core/fusions/fused_moe_permute.py +index bb007b44..98708a5b 100644 +--- a/mindspeed/core/fusions/fused_moe_permute.py ++++ b/mindspeed/core/fusions/fused_moe_permute.py +@@ -100,9 +100,9 @@ def sort_chunks_by_idxs_wrapper(fn): + def moe_alltoall_token_dispatcher_init_wrapper(fn): + @wraps(fn) + def wrapper( +- self, num_local_experts, local_expert_indices, config, model_comm_pgs=None ++ self, num_local_experts, local_expert_indices, config, pg_collection=None + ) -> None: +- fn(self, num_local_experts, local_expert_indices, config, model_comm_pgs) ++ fn(self, num_local_experts, local_expert_indices, config, pg_collection) + # Since fused_sort_chunks_by_index is not currently supported, set self.permute_idx_device to None + self.permute_idx_device = None + input_chunk_idxs = torch.arange( +diff --git a/mindspeed/core/megatron_basic/arguments_basic.py b/mindspeed/core/megatron_basic/arguments_basic.py +index 8ea25b9f..7853bce4 100644 +--- a/mindspeed/core/megatron_basic/arguments_basic.py ++++ b/mindspeed/core/megatron_basic/arguments_basic.py +@@ -113,3 +113,35 @@ def transformer_config_init_wrapper(fn): + fn(self, *args, **known_config) + + return wrapper ++ ++ ++def transformer_config_getattr(self, name): ++ """Resolve MindSpeed extension fields for configs created before patching.""" ++ full_args = vars(get_full_args()) ++ if name in full_args: ++ return full_args[name] ++ raise AttributeError(f"'{type(self).__name__}' object has no attribute '{name}'") ++ ++ ++def transformer_config_init_subclass(cls, **kwargs): ++ mutable_types = (list, dict, set, bytearray) ++ unknown_config = {} ++ full_args = vars(get_full_args()).copy() ++ full_args.update(kwargs) ++ ++ config_key = inspect.signature(cls).parameters ++ for key, value in full_args.items(): ++ if key not in config_key: ++ unknown_config[key] = value ++ ++ for key, value in unknown_config.items(): ++ if not hasattr(cls, key): ++ cls.__annotations__[key] = type(value) ++ value = field(default_factory=value) if callable(value) and not isinstance(value, type) else value ++ if callable(value) and not isinstance(value, type): ++ value = field(default_factory=value) ++ elif type(value) in mutable_types: ++ value = field(default_factory=lambda: value) ++ else: ++ value = value ++ setattr(cls, key, value) +diff --git a/mindspeed/features_manager/megatron_basic/megatron_basic.py b/mindspeed/features_manager/megatron_basic/megatron_basic.py +index 355913e1..41678a03 100644 +--- a/mindspeed/features_manager/megatron_basic/megatron_basic.py ++++ b/mindspeed/features_manager/megatron_basic/megatron_basic.py +@@ -43,8 +43,13 @@ class MegatronBasicFeature(MindSpeedFeature): + + def register_mcore_basic_patches(self, pm, args): + # configuration patches +- from mindspeed.core.megatron_basic.arguments_basic import transformer_config_init_wrapper, transformer_config_post_init_wrapper ++ from mindspeed.core.megatron_basic.arguments_basic import (transformer_config_init_wrapper, ++ transformer_config_getattr, ++ transformer_config_post_init_wrapper, ++ transformer_config_init_subclass) + pm.register_patch("megatron.core.transformer.transformer_config.TransformerConfig.__init__", transformer_config_init_wrapper) ++ pm.register_patch("megatron.core.transformer.transformer_config.TransformerConfig.__init_subclass__", classmethod(transformer_config_init_subclass)) ++ pm.register_patch("megatron.core.transformer.transformer_config.TransformerConfig.__getattr__", transformer_config_getattr, create_dummy=True) + pm.register_patch("megatron.core.transformer.transformer_config.MLATransformerConfig.__init__", transformer_config_init_wrapper) + pm.register_patch("megatron.core.transformer.transformer_config.TransformerConfig.__post_init__", transformer_config_post_init_wrapper) + +diff --git a/mindspeed/patch_utils.py b/mindspeed/patch_utils.py +index a489d58c..d9328555 100644 +--- a/mindspeed/patch_utils.py ++++ b/mindspeed/patch_utils.py +@@ -2,7 +2,7 @@ import importlib + import sys + import types + from typing import List, Dict, Union +- ++import inspect + _MEGATRON_TRAINING_AVAILABLE = None + + +@@ -93,8 +93,10 @@ class Patch: + + def remove_patch(self): + for key, value in sys.modules.copy().items(): +- if 'mindspeed' in key: ++ if 'mindspeed' in key or 'torch.classes' == key: + continue ++ if inspect.isclass(self.orig_module) and hasattr(value, self.orig_module_name.split('.')[-1]): ++ value = getattr(value, self.orig_module_name.split('.')[-1]) + if self.orig_func_name is not None and hasattr(value, self.orig_func_name) \ + and id(getattr(value, self.orig_func_name)) == id(self.final_patch_func): + setattr(value, self.orig_func_name, self.orig_func) +diff --git a/mindspeed/core/fusions/fused_rope.py b/mindspeed/core/fusions/fused_rope.py +index a6f02e07..70f7cb08 100644 +--- a/mindspeed/core/fusions/fused_rope.py ++++ b/mindspeed/core/fusions/fused_rope.py +@@ -126,5 +126,6 @@ def apply_rotary_pos_emb( + freqs, + rotary_interleaved=config.rotary_interleaved, + multi_latent_attention=config.multi_latent_attention, +- mscale=mscale ++ mscale=mscale, ++ cp_group=cp_group + ) +diff --git a/mindspeed/megatron_adaptor.py b/mindspeed/megatron_adaptor.py +index f14b231d..4615590e 100644 +--- a/mindspeed/megatron_adaptor.py ++++ b/mindspeed/megatron_adaptor.py +@@ -57,6 +57,7 @@ def delete_lock_file(): + def repatch(args): + MindSpeedFeaturesManager.remove_patches() + full_args = get_full_args() ++ args = vars(args) + for k, v in args.items(): + setattr(full_args, k, v) + MindSpeedFeaturesManager.apply_features_pre_patches(full_args) +diff --git a/mindspeed/te/pytorch/attention/dot_product_attention/dot_product_attention.py b/mindspeed/te/pytorch/attention/dot_product_attention/dot_product_attention.py +index ac4eabe5..f82a581b 100644 +--- a/mindspeed/te/pytorch/attention/dot_product_attention/dot_product_attention.py ++++ b/mindspeed/te/pytorch/attention/dot_product_attention/dot_product_attention.py +@@ -330,6 +330,8 @@ class DotProductAttention(torch.nn.Module): + inference_params: Any = None, + pad_between_seqs: Optional[bool] = None, + fp8_output: Optional[bool] = False, ++ local_cp_size=None, ++ cp_group=None, + ) -> torch.Tensor: + """ + Dot Product Attention Layer. +@@ -653,6 +655,30 @@ class MindSpeedTEDotProductAttention(DotProductAttention): + packed_seq_params: PackedSeqParams = None, + ): + """Forward.""" ++ packed_seq_kwargs = ( ++ {key: getattr(packed_seq_params, key) for key in self.kept_packed_seq_params} ++ if packed_seq_params is not None ++ else {} ++ ) ++ ++ # Honor the per-call attn_mask_type. Megatron passes AttnMaskType.no_mask for the ++ # (bidirectional) vision tower and AttnMaskType.causal for the causal LM decoder. ++ # The previous code unconditionally built a triu(2048) causal mask and forwarded ++ # config.attention_mask_type, forcing CAUSAL attention even for no_mask -> the ++ # Qwen3-VL vision tower's full attention was computed causally (wrong image embeds). ++ if attn_mask_type == AttnMaskType.no_mask: ++ # Full (bidirectional) attention: no causal mask; per-image segmentation is ++ # handled by cu_seqlens in packed_seq_params (sparse_mode=0 via 'no_mask'). ++ core_attn_out = super().forward( ++ query, ++ key, ++ value, ++ None, ++ attn_mask_type='no_mask', ++ **packed_seq_kwargs, ++ ) ++ return core_attn_out ++ + if ( + attention_mask is None and + self.attn_mask_type == AttnMaskType.causal +@@ -660,12 +686,6 @@ class MindSpeedTEDotProductAttention(DotProductAttention): + self.config.sparse_mode = 2 + attention_mask = get_attention_mask(self.config) + +- packed_seq_kwargs = ( +- {key: getattr(packed_seq_params, key) for key in self.kept_packed_seq_params} +- if packed_seq_params is not None +- else {} +- ) +- + core_attn_out = super().forward( + query, + key, +diff --git a/mindspeed/core/fusions/fused_rope.py b/mindspeed/core/fusions/fused_rope.py +index 70f7cb08..15189c2d 100644 +--- a/mindspeed/core/fusions/fused_rope.py ++++ b/mindspeed/core/fusions/fused_rope.py +@@ -65,7 +65,7 @@ def apply_rotary_pos_emb_bshd( + cos_ = (torch.cos(freqs) * _mscale).to(t.dtype) + sin_ = (torch.sin(freqs) * _mscale).to(t.dtype) + +- if getattr(args, "use_fused_rotary_pos_emb"): ++ if getattr(args, "use_fused_rotary_pos_emb", False): + mode = 1 if rotary_interleaved else 0 + t = npu_rotary_position_embedding(t.contiguous(), cos_, sin_, mode).to(t.dtype) + else: +@@ -82,7 +82,7 @@ def transformer_config_post_init_wrapper(fn): + self.apply_rope_fusion = False + fn(self) + self.apply_rope_fusion = ori_apply_rope_fusion +- if ((getattr(self, "multi_head_latent_attention") or getattr(self, "multi_latent_attention")) ++ if ((getattr(self, "multi_head_latent_attention", False) or getattr(self, "multi_latent_attention", False)) + and self.rope_type == "yarn"): + self.apply_rope_fusion = False + del ori_apply_rope_fusion +diff --git a/mindspeed/core/models/common/embeddings/rotary_pos_embedding.py b/mindspeed/core/models/common/embeddings/rotary_pos_embedding.py +index c0319906..c0741691 100644 +--- a/mindspeed/core/models/common/embeddings/rotary_pos_embedding.py ++++ b/mindspeed/core/models/common/embeddings/rotary_pos_embedding.py +@@ -79,7 +79,7 @@ def apply_rotary_pos_emb_bshd(t: Tensor, freqs: Tensor, rotary_interleaved: bool + cos_ = (torch.cos(freqs) * _mscale).to(t.dtype) + sin_ = (torch.sin(freqs) * _mscale).to(t.dtype) + +- if getattr(args, "use_fused_rotary_pos_emb"): ++ if getattr(args, "use_fused_rotary_pos_emb", False): + mode = 1 if rotary_interleaved else 0 + t = npu_rotary_position_embedding(t.contiguous(), cos_, sin_, mode).to(t.dtype) + else: +diff --git a/mindspeed/features_manager/features_manager.py b/mindspeed/features_manager/features_manager.py +index ba476adc..f5924b98 100644 +--- a/mindspeed/features_manager/features_manager.py ++++ b/mindspeed/features_manager/features_manager.py +@@ -43,7 +43,10 @@ class MindSpeedFeaturesManager: + post_validate_features_args(args=args) # args.x = old_x + """ + for feature in cls.FEATURES_LIST: +- feature.pre_validate_args(args) ++ try: ++ feature.pre_validate_args(args) ++ except AttributeError: ++ pass + + @classmethod + def post_validate_features_args(cls, args): +@@ -54,13 +57,19 @@ class MindSpeedFeaturesManager: + post_validate_features_args(args=args) # args.x = old_x + """ + for feature in cls.FEATURES_LIST: +- feature.post_validate_args(args) ++ try: ++ feature.post_validate_args(args) ++ except AttributeError: ++ pass + + @classmethod + def validate_features_args(cls, args): + """Validate arguments of all features.""" + for feature in cls.FEATURES_LIST: +- feature.validate_args(args) ++ try: ++ feature.validate_args(args) ++ except AttributeError: ++ pass + + @classmethod + def remove_patches(cls): +diff --git a/mindspeed/features_manager/moe/fb_overlap.py b/mindspeed/features_manager/moe/fb_overlap.py +index 6c76d891..e099c51d 100644 +--- a/mindspeed/features_manager/moe/fb_overlap.py ++++ b/mindspeed/features_manager/moe/fb_overlap.py +@@ -16,6 +16,8 @@ class MoEFwdBwdOverlapFeature(MindSpeedFeature): + group.add_argument('--moe-unperm2-mem-optim-swap', action='store_true') + + def validate_args(self, args): ++ if not hasattr(args, 'moe_fb_overlap'): ++ return + self.incompatible_check(args, 'moe_alltoall_overlap_comm') + self.incompatible_check(args, 'overlap_grad_reduce') + self.incompatible_check(args, 'moe_hierarchical_alltoallv') +diff --git a/mindspeed/features_manager/moe/moe_zero_memory.py b/mindspeed/features_manager/moe/moe_zero_memory.py +index 4d95e64e..94f26109 100644 +--- a/mindspeed/features_manager/moe/moe_zero_memory.py ++++ b/mindspeed/features_manager/moe/moe_zero_memory.py +@@ -23,6 +23,8 @@ class MoEZeroMemoryFeature(MindSpeedFeature): + 'in each pp stage.') + + def pre_validate_args(self, args): ++ if not hasattr(args, "moe_zero_memory_num_layers") or not hasattr(args, "moe_zero_memory"): ++ return + #Zero Memory check. + if args.moe_zero_memory_num_layers is not None: + num_layers_per_pipeline_stage = args.num_layers // args.pipeline_model_parallel_size diff --git a/docker/npu_patch/series.conf b/docker/npu_patch/series.conf index 24fd3a586..55c054365 100644 --- a/docker/npu_patch/series.conf +++ b/docker/npu_patch/series.conf @@ -13,3 +13,4 @@ /vllm-workspace/vllm-ascend|vllm-ascend.patch|docker/npu_patch/vllm-ascend.patch /root/Megatron-LM|megatron.patch|docker/npu_patch/megatron.patch /root/Megatron-Bridge|megatron-bridge.patch|docker/npu_patch/megatron-bridge.patch +/root/MindSpeed|mindspeed.patch|docker/npu_patch/mindspeed.patch From 2b7386ebfcc51f25b6aeed063ea52d86578e1d43 Mon Sep 17 00:00:00 2001 From: wangxiaoxin-sherie Date: Mon, 31 Aug 2026 18:38:08 +0800 Subject: [PATCH 05/12] fix(npu): support argparse tuple choices in mindspeed Signed-off-by: wangxiaoxin-sherie --- docker/npu_patch/mindspeed.patch | 12 +++++++++++- 1 file changed, 11 insertions(+), 1 deletion(-) diff --git a/docker/npu_patch/mindspeed.patch b/docker/npu_patch/mindspeed.patch index 0585af488..b796b3f61 100644 --- a/docker/npu_patch/mindspeed.patch +++ b/docker/npu_patch/mindspeed.patch @@ -272,9 +272,19 @@ index 4d95e64e..94f26109 100644 @@ -23,6 +23,8 @@ class MoEZeroMemoryFeature(MindSpeedFeature): 'in each pp stage.') - def pre_validate_args(self, args): + def pre_validate_args(self, args): + if not hasattr(args, "moe_zero_memory_num_layers") or not hasattr(args, "moe_zero_memory"): + return #Zero Memory check. if args.moe_zero_memory_num_layers is not None: num_layers_per_pipeline_stage = args.num_layers // args.pipeline_model_parallel_size + +diff --git a/mindspeed/features_manager/feature.py b/mindspeed/features_manager/feature.py +--- a/mindspeed/features_manager/feature.py ++++ b/mindspeed/features_manager/feature.py +@@ -72,5 +72,7 @@ class MindSpeedFeature: + exist_arg = isinstance(action, argparse.Action) and argument_name in action.option_strings + if exist_arg and action.choices is not None and new_choice not in action.choices: +- action.choices.append(new_choice) ++ # argparse stores choices as a tuple in this image; normalize before extending. ++ action.choices = list(action.choices) + [new_choice] From 3608e570df731999dfc78eacee472fde1424cf25 Mon Sep 17 00:00:00 2001 From: wangxiaoxin-sherie Date: Mon, 31 Aug 2026 18:45:12 +0800 Subject: [PATCH 06/12] fix(npu): align MindSpeed patch context Signed-off-by: wangxiaoxin-sherie --- docker/npu_patch/mindspeed.patch | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/docker/npu_patch/mindspeed.patch b/docker/npu_patch/mindspeed.patch index b796b3f61..68a08b9aa 100644 --- a/docker/npu_patch/mindspeed.patch +++ b/docker/npu_patch/mindspeed.patch @@ -272,7 +272,7 @@ index 4d95e64e..94f26109 100644 @@ -23,6 +23,8 @@ class MoEZeroMemoryFeature(MindSpeedFeature): 'in each pp stage.') - def pre_validate_args(self, args): + def pre_validate_args(self, args): + if not hasattr(args, "moe_zero_memory_num_layers") or not hasattr(args, "moe_zero_memory"): + return #Zero Memory check. @@ -282,7 +282,7 @@ index 4d95e64e..94f26109 100644 diff --git a/mindspeed/features_manager/feature.py b/mindspeed/features_manager/feature.py --- a/mindspeed/features_manager/feature.py +++ b/mindspeed/features_manager/feature.py -@@ -72,5 +72,7 @@ class MindSpeedFeature: +@@ -74,3 +74,4 @@ class MindSpeedFeature: exist_arg = isinstance(action, argparse.Action) and argument_name in action.option_strings if exist_arg and action.choices is not None and new_choice not in action.choices: - action.choices.append(new_choice) From 2c250e4a8f6136a66bd99d749e63c6f44ad6513c Mon Sep 17 00:00:00 2001 From: wangxiaoxin-sherie Date: Mon, 31 Aug 2026 18:51:07 +0800 Subject: [PATCH 07/12] fix(npu): select HCCL for torch dist conversion Signed-off-by: wangxiaoxin-sherie --- tools/convert_hf_to_torch_dist.py | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/tools/convert_hf_to_torch_dist.py b/tools/convert_hf_to_torch_dist.py index 55334bd1b..7b87d507f 100644 --- a/tools/convert_hf_to_torch_dist.py +++ b/tools/convert_hf_to_torch_dist.py @@ -39,6 +39,11 @@ def get_args(): args = parse_args(add_convertion_args) args = set_default_megatron_args(args) + # Megatron's parser only exposes the CUDA backends (nccl/gloo), while + # MindSpeed uses HCCL for model-parallel process groups on Ascend NPU. + if is_npu(): + args.distributed_backend = "hccl" + # set to pass megatron validate_args args.save_interval = 1 args.micro_batch_size = 1 From 177b98344dfde9dd1e40bbc6e60c95848239b729 Mon Sep 17 00:00:00 2001 From: wangxiaoxin-sherie Date: Mon, 31 Aug 2026 18:56:55 +0800 Subject: [PATCH 08/12] fix(npu): disable gloo groups during conversion Signed-off-by: wangxiaoxin-sherie --- tools/convert_hf_to_torch_dist.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/tools/convert_hf_to_torch_dist.py b/tools/convert_hf_to_torch_dist.py index 7b87d507f..9fa9bd064 100644 --- a/tools/convert_hf_to_torch_dist.py +++ b/tools/convert_hf_to_torch_dist.py @@ -43,6 +43,8 @@ def get_args(): # MindSpeed uses HCCL for model-parallel process groups on Ascend NPU. if is_npu(): args.distributed_backend = "hccl" + # Gloo auxiliary groups are CUDA-only in this Megatron/MindSpeed stack. + args.enable_gloo_process_groups = False # set to pass megatron validate_args args.save_interval = 1 From a18b20f7c93248fc87a01d8104e5d8f5f28a5c1d Mon Sep 17 00:00:00 2001 From: wangxiaoxin-sherie Date: Tue, 1 Sep 2026 14:23:42 +0800 Subject: [PATCH 09/12] Revert "fix(npu): disable gloo groups during conversion" This reverts commit 6d7e9691bf26ac3d591b76d49c0ada7d49e8d61e. Signed-off-by: wangxiaoxin-sherie --- tools/convert_hf_to_torch_dist.py | 2 -- 1 file changed, 2 deletions(-) diff --git a/tools/convert_hf_to_torch_dist.py b/tools/convert_hf_to_torch_dist.py index 9fa9bd064..7b87d507f 100644 --- a/tools/convert_hf_to_torch_dist.py +++ b/tools/convert_hf_to_torch_dist.py @@ -43,8 +43,6 @@ def get_args(): # MindSpeed uses HCCL for model-parallel process groups on Ascend NPU. if is_npu(): args.distributed_backend = "hccl" - # Gloo auxiliary groups are CUDA-only in this Megatron/MindSpeed stack. - args.enable_gloo_process_groups = False # set to pass megatron validate_args args.save_interval = 1 From 9843fce325b93d5b81197d5aba657b9f279575b7 Mon Sep 17 00:00:00 2001 From: wangxiaoxin-sherie Date: Tue, 1 Sep 2026 14:23:42 +0800 Subject: [PATCH 10/12] Revert "fix(npu): select HCCL for torch dist conversion" This reverts commit 5e53e29359d3057157bc24f34485cc94b29eff41. Signed-off-by: wangxiaoxin-sherie --- tools/convert_hf_to_torch_dist.py | 5 ----- 1 file changed, 5 deletions(-) diff --git a/tools/convert_hf_to_torch_dist.py b/tools/convert_hf_to_torch_dist.py index 7b87d507f..55334bd1b 100644 --- a/tools/convert_hf_to_torch_dist.py +++ b/tools/convert_hf_to_torch_dist.py @@ -39,11 +39,6 @@ def get_args(): args = parse_args(add_convertion_args) args = set_default_megatron_args(args) - # Megatron's parser only exposes the CUDA backends (nccl/gloo), while - # MindSpeed uses HCCL for model-parallel process groups on Ascend NPU. - if is_npu(): - args.distributed_backend = "hccl" - # set to pass megatron validate_args args.save_interval = 1 args.micro_batch_size = 1 From 535f9d479f7dce9937fa8d505ab26d95f8b50234 Mon Sep 17 00:00:00 2001 From: wangxiaoxin-sherie Date: Tue, 1 Sep 2026 14:23:42 +0800 Subject: [PATCH 11/12] Revert "fix(npu): align MindSpeed patch context" This reverts commit f5ce534bb433816b6091be2081013d68e12b7694. Signed-off-by: wangxiaoxin-sherie --- docker/npu_patch/mindspeed.patch | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/docker/npu_patch/mindspeed.patch b/docker/npu_patch/mindspeed.patch index 68a08b9aa..b796b3f61 100644 --- a/docker/npu_patch/mindspeed.patch +++ b/docker/npu_patch/mindspeed.patch @@ -272,7 +272,7 @@ index 4d95e64e..94f26109 100644 @@ -23,6 +23,8 @@ class MoEZeroMemoryFeature(MindSpeedFeature): 'in each pp stage.') - def pre_validate_args(self, args): + def pre_validate_args(self, args): + if not hasattr(args, "moe_zero_memory_num_layers") or not hasattr(args, "moe_zero_memory"): + return #Zero Memory check. @@ -282,7 +282,7 @@ index 4d95e64e..94f26109 100644 diff --git a/mindspeed/features_manager/feature.py b/mindspeed/features_manager/feature.py --- a/mindspeed/features_manager/feature.py +++ b/mindspeed/features_manager/feature.py -@@ -74,3 +74,4 @@ class MindSpeedFeature: +@@ -72,5 +72,7 @@ class MindSpeedFeature: exist_arg = isinstance(action, argparse.Action) and argument_name in action.option_strings if exist_arg and action.choices is not None and new_choice not in action.choices: - action.choices.append(new_choice) From 5611e375504338a8496b00b1a839dcd21d3619b4 Mon Sep 17 00:00:00 2001 From: wangxiaoxin-sherie Date: Tue, 1 Sep 2026 14:23:42 +0800 Subject: [PATCH 12/12] Revert "fix(npu): support argparse tuple choices in mindspeed" This reverts commit 70d763c950e2f8189887228303e459af68a2547f. Signed-off-by: wangxiaoxin-sherie --- docker/Dockerfile.npu | 8 +------- docker/npu_patch/mindspeed.patch | 12 +----------- 2 files changed, 2 insertions(+), 18 deletions(-) diff --git a/docker/Dockerfile.npu b/docker/Dockerfile.npu index dba692d9f..00cc73f6e 100644 --- a/docker/Dockerfile.npu +++ b/docker/Dockerfile.npu @@ -16,7 +16,6 @@ ARG MBRIDGE_COMMIT=89eb10887887bc74853f89a4de258c0702932a1c ARG SOC_VERSION="ascend910_9391" ARG PIP_INDEX_URL="https://mirrors.tuna.tsinghua.edu.cn/pypi/web/simple" ARG APTMIRROR="" -ARG GITHUB_PROXY="" ENV SOC_VERSION=$SOC_VERSION \ DEBIAN_FRONTEND=noninteractive \ @@ -38,12 +37,7 @@ ENV SOC_VERSION=$SOC_VERSION \ COPY docker/npu_patch /opt/npu_patch COPY docker/patch/latest/megatron.patch /opt/vime_patch/megatron.patch -RUN if [ -n "${GITHUB_PROXY}" ]; then \ - git config --global \ - url."${GITHUB_PROXY%/}/https://github.com/".insteadOf \ - "https://github.com/"; \ - fi && \ - git config --global http.sslVerify false +RUN git config --global http.sslVerify false # System and pip dependencies. RUN if [ -n "$APTMIRROR" ];then sed -i "s@^\(deb.*\)https\?://[a-z0-9.-]*\.ubuntu\.com@\1$APTMIRROR@g" /etc/apt/sources.list; \ diff --git a/docker/npu_patch/mindspeed.patch b/docker/npu_patch/mindspeed.patch index b796b3f61..0585af488 100644 --- a/docker/npu_patch/mindspeed.patch +++ b/docker/npu_patch/mindspeed.patch @@ -272,19 +272,9 @@ index 4d95e64e..94f26109 100644 @@ -23,6 +23,8 @@ class MoEZeroMemoryFeature(MindSpeedFeature): 'in each pp stage.') - def pre_validate_args(self, args): + def pre_validate_args(self, args): + if not hasattr(args, "moe_zero_memory_num_layers") or not hasattr(args, "moe_zero_memory"): + return #Zero Memory check. if args.moe_zero_memory_num_layers is not None: num_layers_per_pipeline_stage = args.num_layers // args.pipeline_model_parallel_size - -diff --git a/mindspeed/features_manager/feature.py b/mindspeed/features_manager/feature.py ---- a/mindspeed/features_manager/feature.py -+++ b/mindspeed/features_manager/feature.py -@@ -72,5 +72,7 @@ class MindSpeedFeature: - exist_arg = isinstance(action, argparse.Action) and argument_name in action.option_strings - if exist_arg and action.choices is not None and new_choice not in action.choices: -- action.choices.append(new_choice) -+ # argparse stores choices as a tuple in this image; normalize before extending. -+ action.choices = list(action.choices) + [new_choice]