Skip to content

CUDA error: an illegal memory access was encountered #54

Description

@quxiaoyang0zero

你好,我在复现research的时候总结在训练一些step后出现 CUDA error: an illegal memory access was encountered 错误,每次都会出现,但是发生的step不相同,请问您有遇到过吗

INFO 04-29 17:17:36 model_runner_base.py:120] Writing input of failed execution to /tmp/err_execute_model_input_20250429-171736.pkl...
WARNING 04-29 17:17:36 model_runner_base.py:143] Failed to pickle inputs of failed execution: CUDA error: an illegal memory access was encountered
WARNING 04-29 17:17:36 model_runner_base.py:143] Compile with TORCH_USE_CUDA_DSA to enable device-side assertions.
WARNING 04-29 17:17:36 model_runner_base.py:143]
Error executing job with overrides: ['algorithm.adv_estimator=grpo', 'algorithm.kl_ctrl.kl_coef=0.001', 'data.train_files=/lpai/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch-main/data/musique/train.parquet', 'data.val_files=/lpai/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch-main/data/musique//test.parquet', 'data.prompt_key=question', 'data.train_batch_size=32', 'data.max_prompt_length=512', 'data.max_response_length=6000', 'data.apply_chat=True', 'data.prompt_template_name=re_search_template_sys', 'actor_rollout_ref.model.path=/lpai/volumes/base-rlhf-ali-sh/quxiaoyang/pretrain_models/Qwen2.5-7B-Instruct', 'actor_rollout_ref.model.enable_gradient_checkpointing=True', 'actor_rollout_ref.model.use_remove_padding=True', 'actor_rollout_ref.actor.optim.lr=1e-6', 'actor_rollout_ref.actor.ppo_mini_batch_size=16', 'actor_rollout_ref.actor.use_dynamic_bsz=True', 'actor_rollout_ref.actor.ppo_max_token_len_per_gpu=13024', 'actor_rollout_ref.actor.use_kl_loss=True', 'actor_rollout_ref.actor.kl_loss_coef=0.001', 'actor_rollout_ref.actor.kl_loss_type=low_var_kl', 'actor_rollout_ref.actor.fsdp_config.param_offload=False', 'actor_rollout_ref.actor.fsdp_config.grad_offload=False', 'actor_rollout_ref.actor.fsdp_config.optimizer_offload=False', 'actor_rollout_ref.rollout.log_prob_max_token_len_per_gpu=26048', 'actor_rollout_ref.rollout.tensor_model_parallel_size=4', 'actor_rollout_ref.rollout.name=vllm_with_search', 'actor_rollout_ref.rollout.gpu_memory_utilization=0.6', 'actor_rollout_ref.rollout.n=4', 'actor_rollout_ref.rollout.search_url=http://172.24.168.194:6006', 'actor_rollout_ref.ref.log_prob_max_token_len_per_gpu=26048', 'actor_rollout_ref.ref.fsdp_config.param_offload=True', 'reward_model.reward_manager=re_search', 'trainer.critic_warmup=0', 'trainer.logger=[console, wandb]', 'trainer.project_name=verl_grpo_4node_32b', 'trainer.experiment_name=verl_grpo_4node_32b_v1', 'trainer.n_gpus_per_node=8', 'trainer.nnodes=4', 'trainer.save_freq=2', 'trainer.test_freq=5', 'trainer.total_epochs=2', 'trainer.default_hdfs_dir=null', 'trainer.default_local_dir=/lpai/volumes/base-rlhf-ali-sh/quxiaoyang/output_model/grpo/lpai_test_0429_7b_6000_n4_at3-2', 'trainer.val_before_train=True', 'trainer.rollout_save_path=']
Traceback (most recent call last):
File "/opt/conda/lib/python3.10/runpy.py", line 196, in _run_module_as_main
return _run_code(code, main_globals, None,
File "/opt/conda/lib/python3.10/runpy.py", line 86, in _run_code
exec(code, run_globals)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/trainer/main_ppo.py", line 146, in
main()
File "/opt/conda/lib/python3.10/site-packages/hydra/main.py", line 94, in decorated_main
_run_hydra(
File "/opt/conda/lib/python3.10/site-packages/hydra/_internal/utils.py", line 394, in _run_hydra
_run_app(
File "/opt/conda/lib/python3.10/site-packages/hydra/_internal/utils.py", line 457, in _run_app
run_and_report(
File "/opt/conda/lib/python3.10/site-packages/hydra/_internal/utils.py", line 223, in run_and_report
raise ex
File "/opt/conda/lib/python3.10/site-packages/hydra/_internal/utils.py", line 220, in run_and_report
return func()
File "/opt/conda/lib/python3.10/site-packages/hydra/_internal/utils.py", line 458, in
lambda: hydra.run(
File "/opt/conda/lib/python3.10/site-packages/hydra/_internal/hydra.py", line 132, in run
_ = ret.return_value
File "/opt/conda/lib/python3.10/site-packages/hydra/core/utils.py", line 260, in return_value
raise self._return_value
File "/opt/conda/lib/python3.10/site-packages/hydra/core/utils.py", line 186, in run_job
ret.return_value = task_function(task_cfg)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/trainer/main_ppo.py", line 29, in main
run_ppo(config)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/trainer/main_ppo.py", line 37, in run_ppo
ray.get(main_task.remote(config, compute_score))
File "/opt/conda/lib/python3.10/site-packages/ray/_private/auto_init_hook.py", line 21, in auto_init_wrapper
return fn(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/ray/_private/client_mode_hook.py", line 103, in wrapper
return func(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/ray/_private/worker.py", line 2667, in get
values, debugger_breakpoint = worker.get_objects(object_refs, timeout=timeout)
File "/opt/conda/lib/python3.10/site-packages/ray/_private/worker.py", line 864, in get_objects
raise value.as_instanceof_cause()
ray.exceptions.RayTaskError(RuntimeError): ray::main_task() (pid=10800, ip=10.80.11.225)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/trainer/main_ppo.py", line 142, in main_task
trainer.fit()
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/trainer/ppo/ray_trainer.py", line 866, in fit
gen_batch_output = self.actor_rollout_wg.generate_sequences(gen_batch)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/single_controller/ray/base.py", line 42, in func
output = ray.get(output)
ray.exceptions.RayTaskError(RuntimeError): ray::WorkerDict.actor_rollout_generate_sequences() (pid=417, ip=10.80.11.200, actor_id=60d1ff12b759290d7d80196502000000, repr=<verl.single_controller.ray.base.WorkerDict object at 0x7fbb4a877be0>)
File "/opt/conda/lib/python3.10/site-packages/vllm/worker/model_runner.py", line 1665, in execute_model
hidden_or_intermediate_states = model_executable(
File "/opt/conda/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1553, in _wrapped_call_impl
return self._call_impl(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1562, in _call_impl
return forward_call(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/vllm/model_executor/models/qwen2.py", line 415, in forward
hidden_states = self.model(input_ids, positions, kv_caches,
File "/opt/conda/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1553, in _wrapped_call_impl
return self._call_impl(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1562, in _call_impl
return forward_call(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/vllm/model_executor/models/qwen2.py", line 288, in forward
hidden_states, residual = layer(
File "/opt/conda/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1553, in _wrapped_call_impl
return self._call_impl(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1562, in _call_impl
return forward_call(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/vllm/model_executor/models/qwen2.py", line 210, in forward
hidden_states = self.self_attn(
File "/opt/conda/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1553, in _wrapped_call_impl
return self._call_impl(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1562, in _call_impl
return forward_call(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/vllm/model_executor/models/qwen2.py", line 157, in forward
attn_output = self.attn(q, k, v, kv_cache, attn_metadata)
File "/opt/conda/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1553, in _wrapped_call_impl
return self._call_impl(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1562, in _call_impl
return forward_call(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/vllm/attention/layer.py", line 100, in forward
return self.impl.forward(query,
File "/opt/conda/lib/python3.10/site-packages/vllm/attention/backends/flash_attn.py", line 586, in forward
output = torch.ops.vllm.unified_flash_attention(
File "/opt/conda/lib/python3.10/site-packages/torch/ops.py", line 1061, in call
return self
._op(*args, **(kwargs or {}))
File "/opt/conda/lib/python3.10/site-packages/torch/_library/custom_ops.py", line 494, in adinplaceorview_impl
return self._opoverload.redispatch(
File "/opt/conda/lib/python3.10/site-packages/torch/ops.py", line 672, in redispatch
return self
._handle.redispatch_boxed(keyset, *args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/torch/_library/custom_ops.py", line 236, in backend_impl
result = self._backend_fns[device_type](*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/vllm/attention/backends/flash_attn.py", line 736, in unified_flash_attention
decode_output = flash_attn_with_kvcache(
File "/opt/conda/lib/python3.10/site-packages/vllm/vllm_flash_attn/flash_attn_interface.py", line 1296, in flash_attn_with_kvcache
out, softmax_lse = torch.ops.vllm_flash_attn_c.fwd_kvcache(
File "/opt/conda/lib/python3.10/site-packages/torch/ops.py", line 1061, in call
return self
._op(*args, **(kwargs or {}))
RuntimeError: CUDA error: an illegal memory access was encountered
Compile with TORCH_USE_CUDA_DSA to enable device-side assertions.

The above exception was the direct cause of the following exception:

ray::WorkerDict.actor_rollout_generate_sequences() (pid=417, ip=10.80.11.200, actor_id=60d1ff12b759290d7d80196502000000, repr=<verl.single_controller.ray.base.WorkerDict object at 0x7fbb4a877be0>)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/workers/fsdp_workers.py", line 490, in generate_sequences
output = self.rollout.generate_sequences(prompts=prompts)
File "/opt/conda/lib/python3.10/site-packages/torch/utils/_contextlib.py", line 116, in decorate_context
return func(*args, **kwargs)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/workers/rollout/vllm_rollout/vllm_rollout.py", line 370, in generate_sequences
outputs = self.inference_engine.generate(
File "/opt/conda/lib/python3.10/site-packages/vllm/utils.py", line 1063, in inner
return fn(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/vllm/entrypoints/llm.py", line 353, in generate
outputs = self._run_engine(use_tqdm=use_tqdm)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/third_party/vllm/vllm_v_0_6_3/llm.py", line 161, in _run_engine
outputs = super()._run_engine(use_tqdm=use_tqdm)
File "/opt/conda/lib/python3.10/site-packages/vllm/entrypoints/llm.py", line 879, in _run_engine
step_outputs = self.llm_engine.step()
File "/opt/conda/lib/python3.10/site-packages/vllm/engine/llm_engine.py", line 1386, in step
outputs = self.model_executor.execute_model(
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/third_party/vllm/vllm_v_0_6_3/spmd_gpu_executor.py", line 163, in execute_model
all_outputs = self.worker.execute_model(execute_model_req=execute_model_req)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/third_party/vllm/vllm_v_0_6_3/worker.py", line 267, in execute_model
return self.model_runner.execute_model(
File "/opt/conda/lib/python3.10/site-packages/torch/utils/_contextlib.py", line 116, in decorate_context
return func(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/vllm/worker/model_runner_base.py", line 146, in _wrapper
raise type(err)(f"Error in model execution: "
RuntimeError: Error in model execution: CUDA error: an illegal memory access was encountered
Compile with TORCH_USE_CUDA_DSA to enable device-side assertions.

During handling of the above exception, another exception occurred:

ray::WorkerDict.actor_rollout_generate_sequences() (pid=417, ip=10.80.11.200, actor_id=60d1ff12b759290d7d80196502000000, repr=<verl.single_controller.ray.base.WorkerDict object at 0x7fbb4a877be0>)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/single_controller/ray/base.py", line 399, in func
return getattr(self.worker_dict[key], name)(*args, **kwargs)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/single_controller/base/decorator.py", line 404, in inner
return func(*args, **kwargs)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/workers/fsdp_workers.py", line 486, in generate_sequences
with self.rollout_sharding_manager:
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/workers/sharding_manager/fsdp_vllm.py", line 120, in exit
torch.cuda.empty_cache()
File "/opt/conda/lib/python3.10/site-packages/torch/cuda/memory.py", line 170, in empty_cache
torch._C._cuda_emptyCache()
RuntimeError: CUDA error: an illegal memory access was encountered
Compile with TORCH_USE_CUDA_DSA to enable device-side assertions.
outputscpu [repeated 15x across cluster]
(WorkerDict pid=11253) active_max_tokens[5097, 5157, 5177, 5164, 5186, 5177, 5172, 5160, 5163, 5210, 5165, 5118, 5145, 5139, 5125, 5145] [repeated 36x across cluster]
/opt/conda/lib/python3.10/site-packages/torch/utils/checkpoint.py:1399: FutureWarning: torch.cpu.amp.autocast(args...) is deprecated. Please use torch.amp.autocast('cpu', args...) instead. [repeated 31x across cluster]
with device_autocast_ctx, torch.cpu.amp.autocast(**cpu_autocast_kwargs), recompute_context: # type: ignore[attr-defined] [repeated 31x across cluster]
ERR cli.py:68 -- ---------------------------------------
ERR cli.py:69 -- Job 'raysubmit_EQdriax8r9e6g3kx' failed
ERR cli.py:70 -- ---------------------------------------
INFO cli.py:83 -- Status message: Job entrypoint command failed with exit code 1, last available logs (truncated to 20,000 chars):
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/workers/sharding_manager/fsdp_vllm.py", line 120, in exit
torch.cuda.empty_cache()
File "/opt/conda/lib/python3.10/site-packages/torch/cuda/memory.py", line 170, in empty_cache
torch._C._cuda_emptyCache()
RuntimeError: CUDA error: an illegal memory access was encountered
Compile with TORCH_USE_CUDA_DSA to enable device-side assertions.
outputscpu [repeated 15x across cluster]

/opt/conda/lib/python3.10/site-packages/torch/utils/checkpoint.py:1399: FutureWarning: torch.cpu.amp.autocast(args...) is deprecated. Please use torch.amp.autocast('cpu', args...) instead. [repeated 31x across cluster]
with device_autocast_ctx, torch.cpu.amp.autocast(**cpu_autocast_kwargs), recompute_context: # type: ignore[attr-defined] [repeated 31x across cluster]

Activity

Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment

Metadata

Metadata

Assignees

No one assigned

    Labels

    No labels
    No labels

    Type

    No type

    Projects

    No projects

      Milestone

      No milestone

      Relationships

      None yet

      Development

      No branches or pull requests

      Issue actions