你好,我在复现research的时候总结在训练一些step后出现 CUDA error: an illegal memory access was encountered 错误,每次都会出现,但是发生的step不相同,请问您有遇到过吗
INFO 04-29 17:17:36 model_runner_base.py:120] Writing input of failed execution to /tmp/err_execute_model_input_20250429-171736.pkl...
WARNING 04-29 17:17:36 model_runner_base.py:143] Failed to pickle inputs of failed execution: CUDA error: an illegal memory access was encountered
WARNING 04-29 17:17:36 model_runner_base.py:143] Compile with TORCH_USE_CUDA_DSA to enable device-side assertions.
WARNING 04-29 17:17:36 model_runner_base.py:143]
Error executing job with overrides: ['algorithm.adv_estimator=grpo', 'algorithm.kl_ctrl.kl_coef=0.001', 'data.train_files=/lpai/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch-main/data/musique/train.parquet', 'data.val_files=/lpai/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch-main/data/musique//test.parquet', 'data.prompt_key=question', 'data.train_batch_size=32', 'data.max_prompt_length=512', 'data.max_response_length=6000', 'data.apply_chat=True', 'data.prompt_template_name=re_search_template_sys', 'actor_rollout_ref.model.path=/lpai/volumes/base-rlhf-ali-sh/quxiaoyang/pretrain_models/Qwen2.5-7B-Instruct', 'actor_rollout_ref.model.enable_gradient_checkpointing=True', 'actor_rollout_ref.model.use_remove_padding=True', 'actor_rollout_ref.actor.optim.lr=1e-6', 'actor_rollout_ref.actor.ppo_mini_batch_size=16', 'actor_rollout_ref.actor.use_dynamic_bsz=True', 'actor_rollout_ref.actor.ppo_max_token_len_per_gpu=13024', 'actor_rollout_ref.actor.use_kl_loss=True', 'actor_rollout_ref.actor.kl_loss_coef=0.001', 'actor_rollout_ref.actor.kl_loss_type=low_var_kl', 'actor_rollout_ref.actor.fsdp_config.param_offload=False', 'actor_rollout_ref.actor.fsdp_config.grad_offload=False', 'actor_rollout_ref.actor.fsdp_config.optimizer_offload=False', 'actor_rollout_ref.rollout.log_prob_max_token_len_per_gpu=26048', 'actor_rollout_ref.rollout.tensor_model_parallel_size=4', 'actor_rollout_ref.rollout.name=vllm_with_search', 'actor_rollout_ref.rollout.gpu_memory_utilization=0.6', 'actor_rollout_ref.rollout.n=4', 'actor_rollout_ref.rollout.search_url=http://172.24.168.194:6006', 'actor_rollout_ref.ref.log_prob_max_token_len_per_gpu=26048', 'actor_rollout_ref.ref.fsdp_config.param_offload=True', 'reward_model.reward_manager=re_search', 'trainer.critic_warmup=0', 'trainer.logger=[console, wandb]', 'trainer.project_name=verl_grpo_4node_32b', 'trainer.experiment_name=verl_grpo_4node_32b_v1', 'trainer.n_gpus_per_node=8', 'trainer.nnodes=4', 'trainer.save_freq=2', 'trainer.test_freq=5', 'trainer.total_epochs=2', 'trainer.default_hdfs_dir=null', 'trainer.default_local_dir=/lpai/volumes/base-rlhf-ali-sh/quxiaoyang/output_model/grpo/lpai_test_0429_7b_6000_n4_at3-2', 'trainer.val_before_train=True', 'trainer.rollout_save_path=']
Traceback (most recent call last):
File "/opt/conda/lib/python3.10/runpy.py", line 196, in _run_module_as_main
return _run_code(code, main_globals, None,
File "/opt/conda/lib/python3.10/runpy.py", line 86, in _run_code
exec(code, run_globals)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/trainer/main_ppo.py", line 146, in
main()
File "/opt/conda/lib/python3.10/site-packages/hydra/main.py", line 94, in decorated_main
_run_hydra(
File "/opt/conda/lib/python3.10/site-packages/hydra/_internal/utils.py", line 394, in _run_hydra
_run_app(
File "/opt/conda/lib/python3.10/site-packages/hydra/_internal/utils.py", line 457, in _run_app
run_and_report(
File "/opt/conda/lib/python3.10/site-packages/hydra/_internal/utils.py", line 223, in run_and_report
raise ex
File "/opt/conda/lib/python3.10/site-packages/hydra/_internal/utils.py", line 220, in run_and_report
return func()
File "/opt/conda/lib/python3.10/site-packages/hydra/_internal/utils.py", line 458, in
lambda: hydra.run(
File "/opt/conda/lib/python3.10/site-packages/hydra/_internal/hydra.py", line 132, in run
_ = ret.return_value
File "/opt/conda/lib/python3.10/site-packages/hydra/core/utils.py", line 260, in return_value
raise self._return_value
File "/opt/conda/lib/python3.10/site-packages/hydra/core/utils.py", line 186, in run_job
ret.return_value = task_function(task_cfg)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/trainer/main_ppo.py", line 29, in main
run_ppo(config)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/trainer/main_ppo.py", line 37, in run_ppo
ray.get(main_task.remote(config, compute_score))
File "/opt/conda/lib/python3.10/site-packages/ray/_private/auto_init_hook.py", line 21, in auto_init_wrapper
return fn(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/ray/_private/client_mode_hook.py", line 103, in wrapper
return func(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/ray/_private/worker.py", line 2667, in get
values, debugger_breakpoint = worker.get_objects(object_refs, timeout=timeout)
File "/opt/conda/lib/python3.10/site-packages/ray/_private/worker.py", line 864, in get_objects
raise value.as_instanceof_cause()
ray.exceptions.RayTaskError(RuntimeError): ray::main_task() (pid=10800, ip=10.80.11.225)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/trainer/main_ppo.py", line 142, in main_task
trainer.fit()
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/trainer/ppo/ray_trainer.py", line 866, in fit
gen_batch_output = self.actor_rollout_wg.generate_sequences(gen_batch)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/single_controller/ray/base.py", line 42, in func
output = ray.get(output)
ray.exceptions.RayTaskError(RuntimeError): ray::WorkerDict.actor_rollout_generate_sequences() (pid=417, ip=10.80.11.200, actor_id=60d1ff12b759290d7d80196502000000, repr=<verl.single_controller.ray.base.WorkerDict object at 0x7fbb4a877be0>)
File "/opt/conda/lib/python3.10/site-packages/vllm/worker/model_runner.py", line 1665, in execute_model
hidden_or_intermediate_states = model_executable(
File "/opt/conda/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1553, in _wrapped_call_impl
return self._call_impl(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1562, in _call_impl
return forward_call(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/vllm/model_executor/models/qwen2.py", line 415, in forward
hidden_states = self.model(input_ids, positions, kv_caches,
File "/opt/conda/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1553, in _wrapped_call_impl
return self._call_impl(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1562, in _call_impl
return forward_call(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/vllm/model_executor/models/qwen2.py", line 288, in forward
hidden_states, residual = layer(
File "/opt/conda/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1553, in _wrapped_call_impl
return self._call_impl(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1562, in _call_impl
return forward_call(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/vllm/model_executor/models/qwen2.py", line 210, in forward
hidden_states = self.self_attn(
File "/opt/conda/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1553, in _wrapped_call_impl
return self._call_impl(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1562, in _call_impl
return forward_call(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/vllm/model_executor/models/qwen2.py", line 157, in forward
attn_output = self.attn(q, k, v, kv_cache, attn_metadata)
File "/opt/conda/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1553, in _wrapped_call_impl
return self._call_impl(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1562, in _call_impl
return forward_call(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/vllm/attention/layer.py", line 100, in forward
return self.impl.forward(query,
File "/opt/conda/lib/python3.10/site-packages/vllm/attention/backends/flash_attn.py", line 586, in forward
output = torch.ops.vllm.unified_flash_attention(
File "/opt/conda/lib/python3.10/site-packages/torch/ops.py", line 1061, in call
return self._op(*args, **(kwargs or {}))
File "/opt/conda/lib/python3.10/site-packages/torch/_library/custom_ops.py", line 494, in adinplaceorview_impl
return self._opoverload.redispatch(
File "/opt/conda/lib/python3.10/site-packages/torch/ops.py", line 672, in redispatch
return self._handle.redispatch_boxed(keyset, *args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/torch/_library/custom_ops.py", line 236, in backend_impl
result = self._backend_fns[device_type](*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/vllm/attention/backends/flash_attn.py", line 736, in unified_flash_attention
decode_output = flash_attn_with_kvcache(
File "/opt/conda/lib/python3.10/site-packages/vllm/vllm_flash_attn/flash_attn_interface.py", line 1296, in flash_attn_with_kvcache
out, softmax_lse = torch.ops.vllm_flash_attn_c.fwd_kvcache(
File "/opt/conda/lib/python3.10/site-packages/torch/ops.py", line 1061, in call
return self._op(*args, **(kwargs or {}))
RuntimeError: CUDA error: an illegal memory access was encountered
Compile with TORCH_USE_CUDA_DSA to enable device-side assertions.
The above exception was the direct cause of the following exception:
ray::WorkerDict.actor_rollout_generate_sequences() (pid=417, ip=10.80.11.200, actor_id=60d1ff12b759290d7d80196502000000, repr=<verl.single_controller.ray.base.WorkerDict object at 0x7fbb4a877be0>)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/workers/fsdp_workers.py", line 490, in generate_sequences
output = self.rollout.generate_sequences(prompts=prompts)
File "/opt/conda/lib/python3.10/site-packages/torch/utils/_contextlib.py", line 116, in decorate_context
return func(*args, **kwargs)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/workers/rollout/vllm_rollout/vllm_rollout.py", line 370, in generate_sequences
outputs = self.inference_engine.generate(
File "/opt/conda/lib/python3.10/site-packages/vllm/utils.py", line 1063, in inner
return fn(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/vllm/entrypoints/llm.py", line 353, in generate
outputs = self._run_engine(use_tqdm=use_tqdm)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/third_party/vllm/vllm_v_0_6_3/llm.py", line 161, in _run_engine
outputs = super()._run_engine(use_tqdm=use_tqdm)
File "/opt/conda/lib/python3.10/site-packages/vllm/entrypoints/llm.py", line 879, in _run_engine
step_outputs = self.llm_engine.step()
File "/opt/conda/lib/python3.10/site-packages/vllm/engine/llm_engine.py", line 1386, in step
outputs = self.model_executor.execute_model(
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/third_party/vllm/vllm_v_0_6_3/spmd_gpu_executor.py", line 163, in execute_model
all_outputs = self.worker.execute_model(execute_model_req=execute_model_req)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/third_party/vllm/vllm_v_0_6_3/worker.py", line 267, in execute_model
return self.model_runner.execute_model(
File "/opt/conda/lib/python3.10/site-packages/torch/utils/_contextlib.py", line 116, in decorate_context
return func(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/vllm/worker/model_runner_base.py", line 146, in _wrapper
raise type(err)(f"Error in model execution: "
RuntimeError: Error in model execution: CUDA error: an illegal memory access was encountered
Compile with TORCH_USE_CUDA_DSA to enable device-side assertions.
During handling of the above exception, another exception occurred:
ray::WorkerDict.actor_rollout_generate_sequences() (pid=417, ip=10.80.11.200, actor_id=60d1ff12b759290d7d80196502000000, repr=<verl.single_controller.ray.base.WorkerDict object at 0x7fbb4a877be0>)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/single_controller/ray/base.py", line 399, in func
return getattr(self.worker_dict[key], name)(*args, **kwargs)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/single_controller/base/decorator.py", line 404, in inner
return func(*args, **kwargs)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/workers/fsdp_workers.py", line 486, in generate_sequences
with self.rollout_sharding_manager:
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/workers/sharding_manager/fsdp_vllm.py", line 120, in exit
torch.cuda.empty_cache()
File "/opt/conda/lib/python3.10/site-packages/torch/cuda/memory.py", line 170, in empty_cache
torch._C._cuda_emptyCache()
RuntimeError: CUDA error: an illegal memory access was encountered
Compile with TORCH_USE_CUDA_DSA to enable device-side assertions.
outputscpu [repeated 15x across cluster]
(WorkerDict pid=11253) active_max_tokens[5097, 5157, 5177, 5164, 5186, 5177, 5172, 5160, 5163, 5210, 5165, 5118, 5145, 5139, 5125, 5145] [repeated 36x across cluster]
/opt/conda/lib/python3.10/site-packages/torch/utils/checkpoint.py:1399: FutureWarning: torch.cpu.amp.autocast(args...) is deprecated. Please use torch.amp.autocast('cpu', args...) instead. [repeated 31x across cluster]
with device_autocast_ctx, torch.cpu.amp.autocast(**cpu_autocast_kwargs), recompute_context: # type: ignore[attr-defined] [repeated 31x across cluster]
ERR cli.py:68 -- ---------------------------------------
ERR cli.py:69 -- Job 'raysubmit_EQdriax8r9e6g3kx' failed
ERR cli.py:70 -- ---------------------------------------
INFO cli.py:83 -- Status message: Job entrypoint command failed with exit code 1, last available logs (truncated to 20,000 chars):
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/workers/sharding_manager/fsdp_vllm.py", line 120, in exit
torch.cuda.empty_cache()
File "/opt/conda/lib/python3.10/site-packages/torch/cuda/memory.py", line 170, in empty_cache
torch._C._cuda_emptyCache()
RuntimeError: CUDA error: an illegal memory access was encountered
Compile with TORCH_USE_CUDA_DSA to enable device-side assertions.
outputscpu [repeated 15x across cluster]
/opt/conda/lib/python3.10/site-packages/torch/utils/checkpoint.py:1399: FutureWarning: torch.cpu.amp.autocast(args...) is deprecated. Please use torch.amp.autocast('cpu', args...) instead. [repeated 31x across cluster]
with device_autocast_ctx, torch.cpu.amp.autocast(**cpu_autocast_kwargs), recompute_context: # type: ignore[attr-defined] [repeated 31x across cluster]
你好,我在复现research的时候总结在训练一些step后出现 CUDA error: an illegal memory access was encountered 错误,每次都会出现,但是发生的step不相同,请问您有遇到过吗
INFO 04-29 17:17:36 model_runner_base.py:120] Writing input of failed execution to /tmp/err_execute_model_input_20250429-171736.pkl...
WARNING 04-29 17:17:36 model_runner_base.py:143] Failed to pickle inputs of failed execution: CUDA error: an illegal memory access was encountered
WARNING 04-29 17:17:36 model_runner_base.py:143] Compile with
TORCH_USE_CUDA_DSAto enable device-side assertions.WARNING 04-29 17:17:36 model_runner_base.py:143]
Error executing job with overrides: ['algorithm.adv_estimator=grpo', 'algorithm.kl_ctrl.kl_coef=0.001', 'data.train_files=/lpai/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch-main/data/musique/train.parquet', 'data.val_files=/lpai/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch-main/data/musique//test.parquet', 'data.prompt_key=question', 'data.train_batch_size=32', 'data.max_prompt_length=512', 'data.max_response_length=6000', 'data.apply_chat=True', 'data.prompt_template_name=re_search_template_sys', 'actor_rollout_ref.model.path=/lpai/volumes/base-rlhf-ali-sh/quxiaoyang/pretrain_models/Qwen2.5-7B-Instruct', 'actor_rollout_ref.model.enable_gradient_checkpointing=True', 'actor_rollout_ref.model.use_remove_padding=True', 'actor_rollout_ref.actor.optim.lr=1e-6', 'actor_rollout_ref.actor.ppo_mini_batch_size=16', 'actor_rollout_ref.actor.use_dynamic_bsz=True', 'actor_rollout_ref.actor.ppo_max_token_len_per_gpu=13024', 'actor_rollout_ref.actor.use_kl_loss=True', 'actor_rollout_ref.actor.kl_loss_coef=0.001', 'actor_rollout_ref.actor.kl_loss_type=low_var_kl', 'actor_rollout_ref.actor.fsdp_config.param_offload=False', 'actor_rollout_ref.actor.fsdp_config.grad_offload=False', 'actor_rollout_ref.actor.fsdp_config.optimizer_offload=False', 'actor_rollout_ref.rollout.log_prob_max_token_len_per_gpu=26048', 'actor_rollout_ref.rollout.tensor_model_parallel_size=4', 'actor_rollout_ref.rollout.name=vllm_with_search', 'actor_rollout_ref.rollout.gpu_memory_utilization=0.6', 'actor_rollout_ref.rollout.n=4', 'actor_rollout_ref.rollout.search_url=http://172.24.168.194:6006', 'actor_rollout_ref.ref.log_prob_max_token_len_per_gpu=26048', 'actor_rollout_ref.ref.fsdp_config.param_offload=True', 'reward_model.reward_manager=re_search', 'trainer.critic_warmup=0', 'trainer.logger=[console, wandb]', 'trainer.project_name=verl_grpo_4node_32b', 'trainer.experiment_name=verl_grpo_4node_32b_v1', 'trainer.n_gpus_per_node=8', 'trainer.nnodes=4', 'trainer.save_freq=2', 'trainer.test_freq=5', 'trainer.total_epochs=2', 'trainer.default_hdfs_dir=null', 'trainer.default_local_dir=/lpai/volumes/base-rlhf-ali-sh/quxiaoyang/output_model/grpo/lpai_test_0429_7b_6000_n4_at3-2', 'trainer.val_before_train=True', 'trainer.rollout_save_path=']
Traceback (most recent call last):
File "/opt/conda/lib/python3.10/runpy.py", line 196, in _run_module_as_main
return _run_code(code, main_globals, None,
File "/opt/conda/lib/python3.10/runpy.py", line 86, in _run_code
exec(code, run_globals)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/trainer/main_ppo.py", line 146, in
main()
File "/opt/conda/lib/python3.10/site-packages/hydra/main.py", line 94, in decorated_main
_run_hydra(
File "/opt/conda/lib/python3.10/site-packages/hydra/_internal/utils.py", line 394, in _run_hydra
_run_app(
File "/opt/conda/lib/python3.10/site-packages/hydra/_internal/utils.py", line 457, in _run_app
run_and_report(
File "/opt/conda/lib/python3.10/site-packages/hydra/_internal/utils.py", line 223, in run_and_report
raise ex
File "/opt/conda/lib/python3.10/site-packages/hydra/_internal/utils.py", line 220, in run_and_report
return func()
File "/opt/conda/lib/python3.10/site-packages/hydra/_internal/utils.py", line 458, in
lambda: hydra.run(
File "/opt/conda/lib/python3.10/site-packages/hydra/_internal/hydra.py", line 132, in run
_ = ret.return_value
File "/opt/conda/lib/python3.10/site-packages/hydra/core/utils.py", line 260, in return_value
raise self._return_value
File "/opt/conda/lib/python3.10/site-packages/hydra/core/utils.py", line 186, in run_job
ret.return_value = task_function(task_cfg)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/trainer/main_ppo.py", line 29, in main
run_ppo(config)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/trainer/main_ppo.py", line 37, in run_ppo
ray.get(main_task.remote(config, compute_score))
File "/opt/conda/lib/python3.10/site-packages/ray/_private/auto_init_hook.py", line 21, in auto_init_wrapper
return fn(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/ray/_private/client_mode_hook.py", line 103, in wrapper
return func(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/ray/_private/worker.py", line 2667, in get
values, debugger_breakpoint = worker.get_objects(object_refs, timeout=timeout)
File "/opt/conda/lib/python3.10/site-packages/ray/_private/worker.py", line 864, in get_objects
raise value.as_instanceof_cause()
ray.exceptions.RayTaskError(RuntimeError): ray::main_task() (pid=10800, ip=10.80.11.225)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/trainer/main_ppo.py", line 142, in main_task
trainer.fit()
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/trainer/ppo/ray_trainer.py", line 866, in fit
gen_batch_output = self.actor_rollout_wg.generate_sequences(gen_batch)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/single_controller/ray/base.py", line 42, in func
output = ray.get(output)
ray.exceptions.RayTaskError(RuntimeError): ray::WorkerDict.actor_rollout_generate_sequences() (pid=417, ip=10.80.11.200, actor_id=60d1ff12b759290d7d80196502000000, repr=<verl.single_controller.ray.base.WorkerDict object at 0x7fbb4a877be0>)
File "/opt/conda/lib/python3.10/site-packages/vllm/worker/model_runner.py", line 1665, in execute_model
hidden_or_intermediate_states = model_executable(
File "/opt/conda/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1553, in _wrapped_call_impl
return self._call_impl(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1562, in _call_impl
return forward_call(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/vllm/model_executor/models/qwen2.py", line 415, in forward
hidden_states = self.model(input_ids, positions, kv_caches,
File "/opt/conda/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1553, in _wrapped_call_impl
return self._call_impl(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1562, in _call_impl
return forward_call(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/vllm/model_executor/models/qwen2.py", line 288, in forward
hidden_states, residual = layer(
File "/opt/conda/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1553, in _wrapped_call_impl
return self._call_impl(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1562, in _call_impl
return forward_call(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/vllm/model_executor/models/qwen2.py", line 210, in forward
hidden_states = self.self_attn(
File "/opt/conda/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1553, in _wrapped_call_impl
return self._call_impl(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1562, in _call_impl
return forward_call(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/vllm/model_executor/models/qwen2.py", line 157, in forward
attn_output = self.attn(q, k, v, kv_cache, attn_metadata)
File "/opt/conda/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1553, in _wrapped_call_impl
return self._call_impl(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/torch/nn/modules/module.py", line 1562, in _call_impl
return forward_call(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/vllm/attention/layer.py", line 100, in forward
return self.impl.forward(query,
File "/opt/conda/lib/python3.10/site-packages/vllm/attention/backends/flash_attn.py", line 586, in forward
output = torch.ops.vllm.unified_flash_attention(
File "/opt/conda/lib/python3.10/site-packages/torch/ops.py", line 1061, in call
return self._op(*args, **(kwargs or {}))
File "/opt/conda/lib/python3.10/site-packages/torch/_library/custom_ops.py", line 494, in adinplaceorview_impl
return self._opoverload.redispatch(
File "/opt/conda/lib/python3.10/site-packages/torch/ops.py", line 672, in redispatch
return self._handle.redispatch_boxed(keyset, *args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/torch/_library/custom_ops.py", line 236, in backend_impl
result = self._backend_fns[device_type](*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/vllm/attention/backends/flash_attn.py", line 736, in unified_flash_attention
decode_output = flash_attn_with_kvcache(
File "/opt/conda/lib/python3.10/site-packages/vllm/vllm_flash_attn/flash_attn_interface.py", line 1296, in flash_attn_with_kvcache
out, softmax_lse = torch.ops.vllm_flash_attn_c.fwd_kvcache(
File "/opt/conda/lib/python3.10/site-packages/torch/ops.py", line 1061, in call
return self._op(*args, **(kwargs or {}))
RuntimeError: CUDA error: an illegal memory access was encountered
Compile with
TORCH_USE_CUDA_DSAto enable device-side assertions.The above exception was the direct cause of the following exception:
ray::WorkerDict.actor_rollout_generate_sequences() (pid=417, ip=10.80.11.200, actor_id=60d1ff12b759290d7d80196502000000, repr=<verl.single_controller.ray.base.WorkerDict object at 0x7fbb4a877be0>)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/workers/fsdp_workers.py", line 490, in generate_sequences
output = self.rollout.generate_sequences(prompts=prompts)
File "/opt/conda/lib/python3.10/site-packages/torch/utils/_contextlib.py", line 116, in decorate_context
return func(*args, **kwargs)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/workers/rollout/vllm_rollout/vllm_rollout.py", line 370, in generate_sequences
outputs = self.inference_engine.generate(
File "/opt/conda/lib/python3.10/site-packages/vllm/utils.py", line 1063, in inner
return fn(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/vllm/entrypoints/llm.py", line 353, in generate
outputs = self._run_engine(use_tqdm=use_tqdm)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/third_party/vllm/vllm_v_0_6_3/llm.py", line 161, in _run_engine
outputs = super()._run_engine(use_tqdm=use_tqdm)
File "/opt/conda/lib/python3.10/site-packages/vllm/entrypoints/llm.py", line 879, in _run_engine
step_outputs = self.llm_engine.step()
File "/opt/conda/lib/python3.10/site-packages/vllm/engine/llm_engine.py", line 1386, in step
outputs = self.model_executor.execute_model(
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/third_party/vllm/vllm_v_0_6_3/spmd_gpu_executor.py", line 163, in execute_model
all_outputs = self.worker.execute_model(execute_model_req=execute_model_req)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/third_party/vllm/vllm_v_0_6_3/worker.py", line 267, in execute_model
return self.model_runner.execute_model(
File "/opt/conda/lib/python3.10/site-packages/torch/utils/_contextlib.py", line 116, in decorate_context
return func(*args, **kwargs)
File "/opt/conda/lib/python3.10/site-packages/vllm/worker/model_runner_base.py", line 146, in _wrapper
raise type(err)(f"Error in model execution: "
RuntimeError: Error in model execution: CUDA error: an illegal memory access was encountered
Compile with
TORCH_USE_CUDA_DSAto enable device-side assertions.During handling of the above exception, another exception occurred:
ray::WorkerDict.actor_rollout_generate_sequences() (pid=417, ip=10.80.11.200, actor_id=60d1ff12b759290d7d80196502000000, repr=<verl.single_controller.ray.base.WorkerDict object at 0x7fbb4a877be0>)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/single_controller/ray/base.py", line 399, in func
return getattr(self.worker_dict[key], name)(*args, **kwargs)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/single_controller/base/decorator.py", line 404, in inner
return func(*args, **kwargs)
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/workers/fsdp_workers.py", line 486, in generate_sequences
with self.rollout_sharding_manager:
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/workers/sharding_manager/fsdp_vllm.py", line 120, in exit
torch.cuda.empty_cache()
File "/opt/conda/lib/python3.10/site-packages/torch/cuda/memory.py", line 170, in empty_cache
torch._C._cuda_emptyCache()
RuntimeError: CUDA error: an illegal memory access was encountered
Compile with
TORCH_USE_CUDA_DSAto enable device-side assertions.outputscpu [repeated 15x across cluster]
(WorkerDict pid=11253) active_max_tokens[5097, 5157, 5177, 5164, 5186, 5177, 5172, 5160, 5163, 5210, 5165, 5118, 5145, 5139, 5125, 5145] [repeated 36x across cluster]
/opt/conda/lib/python3.10/site-packages/torch/utils/checkpoint.py:1399: FutureWarning:
torch.cpu.amp.autocast(args...)is deprecated. Please usetorch.amp.autocast('cpu', args...)instead. [repeated 31x across cluster]with device_autocast_ctx, torch.cpu.amp.autocast(**cpu_autocast_kwargs), recompute_context: # type: ignore[attr-defined] [repeated 31x across cluster]
ERR cli.py:68 -- ---------------------------------------
ERR cli.py:69 -- Job 'raysubmit_EQdriax8r9e6g3kx' failed
ERR cli.py:70 -- ---------------------------------------
INFO cli.py:83 -- Status message: Job entrypoint command failed with exit code 1, last available logs (truncated to 20,000 chars):
File "/mnt/volumes/base-rlhf-ali-sh/quxiaoyang/tools/ReSearch/src/verl/workers/sharding_manager/fsdp_vllm.py", line 120, in exit
torch.cuda.empty_cache()
File "/opt/conda/lib/python3.10/site-packages/torch/cuda/memory.py", line 170, in empty_cache
torch._C._cuda_emptyCache()
RuntimeError: CUDA error: an illegal memory access was encountered
Compile with
TORCH_USE_CUDA_DSAto enable device-side assertions.outputscpu [repeated 15x across cluster]
/opt/conda/lib/python3.10/site-packages/torch/utils/checkpoint.py:1399: FutureWarning:
torch.cpu.amp.autocast(args...)is deprecated. Please usetorch.amp.autocast('cpu', args...)instead. [repeated 31x across cluster]with device_autocast_ctx, torch.cpu.amp.autocast(**cpu_autocast_kwargs), recompute_context: # type: ignore[attr-defined] [repeated 31x across cluster]