[Bugfix] fix the oom when chunkprefill with long context like 64k

phyde682 · phyde682 · commit 4e39ecd2e60e · 2025-08-11T20:05:52.000+08:00
Signed-off-by: haojiangzheng &lt;justineric096@gmail.com&gt;
diff --git a/vllm_ascend/worker/model_runner_v1.py b/vllm_ascend/worker/model_runner_v1.py
@@ -829,7 +829,7 @@ def get_supported_tasks(self) -> "tuple[SupportedTask, ...]":
     def _make_attention_mask(self, seq_lens, query_lens, position,
                              attn_state) -> torch.Tensor:
         # Chunk Prefill situation.
-        if attn_state == AscendAttentionState.ChunkedPrefill:
+        if attn_state == AscendAttentionState.ChunkedPrefill and not self.vllm_config.model_config.use_mla:
             return self.attn_mask_builder.get_splitfuse_attn_mask(
                 seq_lens, query_lens, position, self.dtype, self.device)
         # Prefill without cache situation.