Component
Verifier / IR semantics (lib/PTO/IR)
Description
提单:pto_low_level_loop_fusion — interstage setup 引用前序 loop result 后 erase abort
1. 问题来源
- pass:
pto_low_level_loop_fusion (-pto-low-level-loop-fusion)
- 涉及文件:
$PTOAS_ROOT/lib/PTO/Transforms/TileFusion/PTOLowLevelLoopFusion.cpp
- 关键逻辑:
fuseStageRun → setupOp->moveBefore(fusedOuterLoop) → stage.getOuterLoop().erase()
- 发现途径: 静态复核可疑点 4 + 手工 PoC 差分实证
- 确认时间: 2026-09-18
2. 为什么是 bug
collectStageRunFrom 允许 stage 之间的 pure op 作为 setupOps(isInterstageSetupOp)。
这些 op 可以合法引用前一个 scf.for 的 result。
融合时:
buildFusedLoopNestAtLevel 只把原 loop result 记进 IRMapping(mapFusedLoopResults)
- 没有对 IR 上的外部 use 做
replaceAllUsesWith
- 随后
setupOps 被 moveBefore(fusedOuterLoop),仍指向旧 loop result
erase() 旧 loop → LLVM ERROR: operation destroyed but still has uses(abort)
3. 实证
|
无 pass |
挂 pass |
bug.pto(interstage arith.addi 使用前序 loop result) |
exit 0 |
exit 134,destroyed but still has uses |
control.pto(同一位置 addi 不依赖 loop result) |
exit 0 |
exit 0,正常融合 |
bash test/pto_low_level_loop_fusion_bug_1/reproduce.sh
# 预期: 差分复现成功
4. 修复建议
在 erase 旧 loop 之前,把每个原 scf.for 的 result 对所有外部 use(含 setupOps)
replaceAllUsesWith 到 fused loop 对应 result;或拒绝收集「使用前序 stage loop result」的 setup op。
Reproduction (minimal)
bug.pto
module attributes {pto.backend = "vpto", pto.kernel_kind = #pto.kernel_kind<vector>, pto.target_arch = "a5"} {
func.func @fusion_backend_lifecycle_level3(%arg0: !pto.ptr<f32, gm>) {
%c24576_i64 = arith.constant 24576 : i64
%c20480_i64 = arith.constant 20480 : i64
%c16384_i64 = arith.constant 16384 : i64
%c12288_i64 = arith.constant 12288 : i64
%c8192_i64 = arith.constant 8192 : i64
%c4096_i64 = arith.constant 4096 : i64
%c0_i64 = arith.constant 0 : i64
%c0 = arith.constant 0 : index
%c1 = arith.constant 1 : index
%c32 = arith.constant 32 : index
%c1024 = arith.constant 1024 : index
%0 = pto.make_tensor_view %arg0, shape = [%c1, %c1, %c1, %c32, %c32], strides = [%c1024, %c1024, %c1024, %c32, %c1] {layout = #pto.layout<nd>, pto.inferred_layout = true} : !pto.tensor_view<1x1x1x32x32xf32>
%1 = pto.partition_view %0, offsets = [%c0, %c0, %c0, %c0, %c0], sizes = [%c1, %c1, %c1, %c32, %c32] : !pto.tensor_view<1x1x1x32x32xf32>
%2 = pto.alloc_tile addr = %c0_i64 : !pto.tile_buf<vec, 32x32xf32>
%3 = pto.alloc_tile addr = %c4096_i64 : !pto.tile_buf<vec, 32x32xf32>
%4 = pto.alloc_tile addr = %c8192_i64 : !pto.tile_buf<vec, 32x32xf32>
%5 = pto.alloc_tile addr = %c12288_i64 : !pto.tile_buf<vec, 32x32xf32>
%6 = pto.fusion_region {
%11 = pto.alloc_tile addr = %c16384_i64 : !pto.tile_buf<vec, 32x32xf32>
%12 = pto.alloc_tile addr = %c20480_i64 : !pto.tile_buf<vec, 32x32xf32>
%13 = pto.alloc_tile addr = %c24576_i64 : !pto.tile_buf<vec, 32x32xf32>
%14 = pto.tile_buf_addr %2 : !pto.tile_buf<vec, 32x32xf32> -> !pto.ptr<f32, ub>
%15 = pto.tile_buf_addr %3 : !pto.tile_buf<vec, 32x32xf32> -> !pto.ptr<f32, ub>
%16 = pto.tile_buf_addr %11 : !pto.tile_buf<vec, 32x32xf32> -> !pto.ptr<f32, ub>
%c32_5 = arith.constant 32 : index
%c32_6 = arith.constant 32 : index
%17 = arith.muli %c32_5, %c32_6 : index
%c0_7 = arith.constant 0 : index
%c64 = arith.constant 64 : index
%18 = scf.for %arg1 = %c0_7 to %17 step %c64 iter_args(%arg2 = %17) -> (index) {
%29 = arith.index_cast %arg2 : index to i32
%mask, %scalar_out = pto.plt_b32 %29 : i32 -> !pto.mask<b32>, i32
%30 = arith.index_cast %scalar_out : i32 to index
%result = pto.vlds %14[%arg1] : !pto.ptr<f32, ub> -> !pto.vreg<64xf32>
%result_16 = pto.vlds %15[%arg1] : !pto.ptr<f32, ub> -> !pto.vreg<64xf32>
%31 = pto.vadd %result, %result_16, %mask : !pto.vreg<64xf32>, !pto.vreg<64xf32>, !pto.mask<b32> -> !pto.vreg<64xf32>
pto.vsts %31, %16[%arg1], %mask : !pto.vreg<64xf32>, !pto.ptr<f32, ub>, !pto.mask<b32>
scf.yield %30 : index
}
%setup_use = arith.addi %18, %c0_7 : index
%19 = pto.tile_buf_addr %4 : !pto.tile_buf<vec, 32x32xf32> -> !pto.ptr<f32, ub>
%20 = pto.tile_buf_addr %5 : !pto.tile_buf<vec, 32x32xf32> -> !pto.ptr<f32, ub>
%21 = pto.tile_buf_addr %12 : !pto.tile_buf<vec, 32x32xf32> -> !pto.ptr<f32, ub>
%c32_8 = arith.constant 32 : index
%c32_9 = arith.constant 32 : index
%22 = arith.muli %c32_8, %c32_9 : index
%c0_10 = arith.constant 0 : index
%c64_11 = arith.constant 64 : index
%23 = scf.for %arg1 = %c0_10 to %22 step %c64_11 iter_args(%arg2 = %22) -> (index) {
%29 = arith.index_cast %arg2 : index to i32
%mask, %scalar_out = pto.plt_b32 %29 : i32 -> !pto.mask<b32>, i32
%30 = arith.index_cast %scalar_out : i32 to index
%result = pto.vlds %19[%arg1] : !pto.ptr<f32, ub> -> !pto.vreg<64xf32>
%result_16 = pto.vlds %20[%arg1] : !pto.ptr<f32, ub> -> !pto.vreg<64xf32>
%31 = pto.vadd %result, %result_16, %mask : !pto.vreg<64xf32>, !pto.vreg<64xf32>, !pto.mask<b32> -> !pto.vreg<64xf32>
pto.vsts %31, %21[%arg1], %mask : !pto.vreg<64xf32>, !pto.ptr<f32, ub>, !pto.mask<b32>
scf.yield %30 : index
}
%24 = pto.tile_buf_addr %11 : !pto.tile_buf<vec, 32x32xf32> -> !pto.ptr<f32, ub>
%25 = pto.tile_buf_addr %12 : !pto.tile_buf<vec, 32x32xf32> -> !pto.ptr<f32, ub>
%26 = pto.tile_buf_addr %13 : !pto.tile_buf<vec, 32x32xf32> -> !pto.ptr<f32, ub>
%c32_12 = arith.constant 32 : index
%c32_13 = arith.constant 32 : index
%27 = arith.muli %c32_12, %c32_13 : index
%c0_14 = arith.constant 0 : index
%c64_15 = arith.constant 64 : index
%28 = scf.for %arg1 = %c0_14 to %27 step %c64_15 iter_args(%arg2 = %27) -> (index) {
%29 = arith.index_cast %arg2 : index to i32
%mask, %scalar_out = pto.plt_b32 %29 : i32 -> !pto.mask<b32>, i32
%30 = arith.index_cast %scalar_out : i32 to index
%result = pto.vlds %24[%arg1] : !pto.ptr<f32, ub> -> !pto.vreg<64xf32>
%result_16 = pto.vlds %25[%arg1] : !pto.ptr<f32, ub> -> !pto.vreg<64xf32>
%31 = pto.vadd %result, %result_16, %mask : !pto.vreg<64xf32>, !pto.vreg<64xf32>, !pto.mask<b32> -> !pto.vreg<64xf32>
pto.vsts %31, %26[%arg1], %mask : !pto.vreg<64xf32>, !pto.ptr<f32, ub>, !pto.mask<b32>
scf.yield %30 : index
}
pto.yield(%13) : (!pto.tile_buf<vec, 32x32xf32>) -> ()
} {pto.fusion.group_id = 0 : i64} : !pto.tile_buf<vec, 32x32xf32>
%7 = builtin.unrealized_conversion_cast %1 : !pto.partition_tensor_view<1x1x1x32x32xf32> to !pto.tensor_view<1x1x1x32x32xf32>
%c32_0 = arith.constant 32 : index
%c32_1 = arith.constant 32 : index
%c4 = arith.constant 4 : index
%8 = arith.muli %c32_1, %c4 : index
%9 = pto.tile_buf_addr %6 : !pto.tile_buf<vec, 32x32xf32> -> !pto.ptr<f32, ub>
%10 = pto.tensor_view_addr %7 : !pto.tensor_view<1x1x1x32x32xf32> -> !pto.ptr<f32, gm>
%c0_2 = arith.constant 0 : index
%c1_3 = arith.constant 1 : index
%c1_4 = arith.constant 1 : index
scf.for %arg1 = %c0_2 to %c1_3 step %c1_4 {
%c1024_5 = arith.constant 1024 : index
%11 = arith.muli %arg1, %c1024_5 : index
%12 = pto.addptr %9, %11 : <f32, ub> -> <f32, ub>
%c1024_6 = arith.constant 1024 : index
%13 = arith.muli %arg1, %c1024_6 : index
%14 = pto.addptr %10, %13 : <f32, gm> -> <f32, gm>
%c32_i64 = arith.constant 32 : i64
%c128_i64 = arith.constant 128 : i64
%c128_i64_7 = arith.constant 128 : i64
%15 = arith.index_cast %8 : index to i64
%c0_i64_8 = arith.constant 0 : i64
pto.mte_ub_gm %12, %14, %15 nburst(%c32_i64, %c128_i64, %c128_i64_7) l2_cache_ctl(%c0_i64_8) {operandSegmentSizes = array<i32: 1, 1, 1, 1, 1, 1, 1, 0, 0, 0>} : !pto.ptr<f32, ub>, !pto.ptr<f32, gm>, i64, i64, i64, i64, i64
}
return
}
}
control.pto
module attributes {pto.backend = "vpto", pto.kernel_kind = #pto.kernel_kind<vector>, pto.target_arch = "a5"} {
func.func @fusion_backend_lifecycle_level3(%arg0: !pto.ptr<f32, gm>) {
%c24576_i64 = arith.constant 24576 : i64
%c20480_i64 = arith.constant 20480 : i64
%c16384_i64 = arith.constant 16384 : i64
%c12288_i64 = arith.constant 12288 : i64
%c8192_i64 = arith.constant 8192 : i64
%c4096_i64 = arith.constant 4096 : i64
%c0_i64 = arith.constant 0 : i64
%c0 = arith.constant 0 : index
%c1 = arith.constant 1 : index
%c32 = arith.constant 32 : index
%c1024 = arith.constant 1024 : index
%0 = pto.make_tensor_view %arg0, shape = [%c1, %c1, %c1, %c32, %c32], strides = [%c1024, %c1024, %c1024, %c32, %c1] {layout = #pto.layout<nd>, pto.inferred_layout = true} : !pto.tensor_view<1x1x1x32x32xf32>
%1 = pto.partition_view %0, offsets = [%c0, %c0, %c0, %c0, %c0], sizes = [%c1, %c1, %c1, %c32, %c32] : !pto.tensor_view<1x1x1x32x32xf32>
%2 = pto.alloc_tile addr = %c0_i64 : !pto.tile_buf<vec, 32x32xf32>
%3 = pto.alloc_tile addr = %c4096_i64 : !pto.tile_buf<vec, 32x32xf32>
%4 = pto.alloc_tile addr = %c8192_i64 : !pto.tile_buf<vec, 32x32xf32>
%5 = pto.alloc_tile addr = %c12288_i64 : !pto.tile_buf<vec, 32x32xf32>
%6 = pto.fusion_region {
%11 = pto.alloc_tile addr = %c16384_i64 : !pto.tile_buf<vec, 32x32xf32>
%12 = pto.alloc_tile addr = %c20480_i64 : !pto.tile_buf<vec, 32x32xf32>
%13 = pto.alloc_tile addr = %c24576_i64 : !pto.tile_buf<vec, 32x32xf32>
%14 = pto.tile_buf_addr %2 : !pto.tile_buf<vec, 32x32xf32> -> !pto.ptr<f32, ub>
%15 = pto.tile_buf_addr %3 : !pto.tile_buf<vec, 32x32xf32> -> !pto.ptr<f32, ub>
%16 = pto.tile_buf_addr %11 : !pto.tile_buf<vec, 32x32xf32> -> !pto.ptr<f32, ub>
%c32_5 = arith.constant 32 : index
%c32_6 = arith.constant 32 : index
%17 = arith.muli %c32_5, %c32_6 : index
%c0_7 = arith.constant 0 : index
%c64 = arith.constant 64 : index
%18 = scf.for %arg1 = %c0_7 to %17 step %c64 iter_args(%arg2 = %17) -> (index) {
%29 = arith.index_cast %arg2 : index to i32
%mask, %scalar_out = pto.plt_b32 %29 : i32 -> !pto.mask<b32>, i32
%30 = arith.index_cast %scalar_out : i32 to index
%result = pto.vlds %14[%arg1] : !pto.ptr<f32, ub> -> !pto.vreg<64xf32>
%result_16 = pto.vlds %15[%arg1] : !pto.ptr<f32, ub> -> !pto.vreg<64xf32>
%31 = pto.vadd %result, %result_16, %mask : !pto.vreg<64xf32>, !pto.vreg<64xf32>, !pto.mask<b32> -> !pto.vreg<64xf32>
pto.vsts %31, %16[%arg1], %mask : !pto.vreg<64xf32>, !pto.ptr<f32, ub>, !pto.mask<b32>
scf.yield %30 : index
}
%setup_use = arith.addi %c0_7, %c0_7 : index
%19 = pto.tile_buf_addr %4 : !pto.tile_buf<vec, 32x32xf32> -> !pto.ptr<f32, ub>
%20 = pto.tile_buf_addr %5 : !pto.tile_buf<vec, 32x32xf32> -> !pto.ptr<f32, ub>
%21 = pto.tile_buf_addr %12 : !pto.tile_buf<vec, 32x32xf32> -> !pto.ptr<f32, ub>
%c32_8 = arith.constant 32 : index
%c32_9 = arith.constant 32 : index
%22 = arith.muli %c32_8, %c32_9 : index
%c0_10 = arith.constant 0 : index
%c64_11 = arith.constant 64 : index
%23 = scf.for %arg1 = %c0_10 to %22 step %c64_11 iter_args(%arg2 = %22) -> (index) {
%29 = arith.index_cast %arg2 : index to i32
%mask, %scalar_out = pto.plt_b32 %29 : i32 -> !pto.mask<b32>, i32
%30 = arith.index_cast %scalar_out : i32 to index
%result = pto.vlds %19[%arg1] : !pto.ptr<f32, ub> -> !pto.vreg<64xf32>
%result_16 = pto.vlds %20[%arg1] : !pto.ptr<f32, ub> -> !pto.vreg<64xf32>
%31 = pto.vadd %result, %result_16, %mask : !pto.vreg<64xf32>, !pto.vreg<64xf32>, !pto.mask<b32> -> !pto.vreg<64xf32>
pto.vsts %31, %21[%arg1], %mask : !pto.vreg<64xf32>, !pto.ptr<f32, ub>, !pto.mask<b32>
scf.yield %30 : index
}
%24 = pto.tile_buf_addr %11 : !pto.tile_buf<vec, 32x32xf32> -> !pto.ptr<f32, ub>
%25 = pto.tile_buf_addr %12 : !pto.tile_buf<vec, 32x32xf32> -> !pto.ptr<f32, ub>
%26 = pto.tile_buf_addr %13 : !pto.tile_buf<vec, 32x32xf32> -> !pto.ptr<f32, ub>
%c32_12 = arith.constant 32 : index
%c32_13 = arith.constant 32 : index
%27 = arith.muli %c32_12, %c32_13 : index
%c0_14 = arith.constant 0 : index
%c64_15 = arith.constant 64 : index
%28 = scf.for %arg1 = %c0_14 to %27 step %c64_15 iter_args(%arg2 = %27) -> (index) {
%29 = arith.index_cast %arg2 : index to i32
%mask, %scalar_out = pto.plt_b32 %29 : i32 -> !pto.mask<b32>, i32
%30 = arith.index_cast %scalar_out : i32 to index
%result = pto.vlds %24[%arg1] : !pto.ptr<f32, ub> -> !pto.vreg<64xf32>
%result_16 = pto.vlds %25[%arg1] : !pto.ptr<f32, ub> -> !pto.vreg<64xf32>
%31 = pto.vadd %result, %result_16, %mask : !pto.vreg<64xf32>, !pto.vreg<64xf32>, !pto.mask<b32> -> !pto.vreg<64xf32>
pto.vsts %31, %26[%arg1], %mask : !pto.vreg<64xf32>, !pto.ptr<f32, ub>, !pto.mask<b32>
scf.yield %30 : index
}
pto.yield(%13) : (!pto.tile_buf<vec, 32x32xf32>) -> ()
} {pto.fusion.group_id = 0 : i64} : !pto.tile_buf<vec, 32x32xf32>
%7 = builtin.unrealized_conversion_cast %1 : !pto.partition_tensor_view<1x1x1x32x32xf32> to !pto.tensor_view<1x1x1x32x32xf32>
%c32_0 = arith.constant 32 : index
%c32_1 = arith.constant 32 : index
%c4 = arith.constant 4 : index
%8 = arith.muli %c32_1, %c4 : index
%9 = pto.tile_buf_addr %6 : !pto.tile_buf<vec, 32x32xf32> -> !pto.ptr<f32, ub>
%10 = pto.tensor_view_addr %7 : !pto.tensor_view<1x1x1x32x32xf32> -> !pto.ptr<f32, gm>
%c0_2 = arith.constant 0 : index
%c1_3 = arith.constant 1 : index
%c1_4 = arith.constant 1 : index
scf.for %arg1 = %c0_2 to %c1_3 step %c1_4 {
%c1024_5 = arith.constant 1024 : index
%11 = arith.muli %arg1, %c1024_5 : index
%12 = pto.addptr %9, %11 : <f32, ub> -> <f32, ub>
%c1024_6 = arith.constant 1024 : index
%13 = arith.muli %arg1, %c1024_6 : index
%14 = pto.addptr %10, %13 : <f32, gm> -> <f32, gm>
%c32_i64 = arith.constant 32 : i64
%c128_i64 = arith.constant 128 : i64
%c128_i64_7 = arith.constant 128 : i64
%15 = arith.index_cast %8 : index to i64
%c0_i64_8 = arith.constant 0 : i64
pto.mte_ub_gm %12, %14, %15 nburst(%c32_i64, %c128_i64, %c128_i64_7) l2_cache_ctl(%c0_i64_8) {operandSegmentSizes = array<i32: 1, 1, 1, 1, 1, 1, 1, 0, 0, 0>} : !pto.ptr<f32, ub>, !pto.ptr<f32, gm>, i64, i64, i64, i64, i64
}
return
}
}
两个文件差异:
reproduce.sh
#!/usr/bin/env bash
# 一键复现: pto_low_level_loop_fusion / setupOps 引用前序 loop result 后 erase abort
set -u
HERE="$(cd "$(dirname "$0")" && pwd)"
REPO="${PTOAS_ROOT:-$HOME/code/pto/PTOAS}"
PASS='-pto-low-level-loop-fusion'
BIN="${1:-}"
if [ -z "$BIN" ]; then
for cand in "$REPO/build/tools/pto-test-opt/pto-test-opt" \
"$REPO/build-coverage/tools/pto-test-opt/pto-test-opt"; do
[ -x "$cand" ] && BIN="$cand" && break
done
fi
if [ -z "$BIN" ] || [ ! -x "$BIN" ]; then
echo "[!] 未找到 pto-test-opt" >&2
exit 3
fi
echo "[i] 使用二进制: $BIN"
run_one() {
local name="$1"
"$BIN" $PASS "$HERE/${name}.pto" >"$HERE/${name}.out.pto" 2>"$HERE/${name}.stderr.txt"
echo "$?"
}
ctl=$(run_one control)
bug=$(run_one bug)
echo "[control] exit=$ctl (预期 0)"
echo "[bug] exit=$bug (预期非 0,operation destroyed but still has uses / abort)"
echo "--- bug.stderr ---"
head -40 "$HERE/bug.stderr.txt"
if [ "$ctl" -eq 0 ] && [ "$bug" -ne 0 ]; then
if grep -qE 'destroyed but still has uses|LLVM ERROR' "$HERE/bug.stderr.txt"; then
echo "[OK] 差分复现成功"
exit 1
fi
fi
if [ "$ctl" -eq 0 ] && [ "$bug" -eq 0 ]; then
echo "[OK] 未复现(可能已修复)"
exit 0
fi
echo "[!] 结果异常"
exit 2
Expected behavior
编译通过
Actual behavior / error logs
pto-test-opt -pto-low-level-loop-fusion bug.pto
bug.pto:32:13: error: 'scf.for' op operation destroyed but still has uses
%18 = scf.for %arg1 = %c0_7 to %17 step %c64 iter_args(%arg2 = %17) -> (index) {
^
bug.pto:32:13: note: see current operation:
%0 = "scf.for"(<<UNKNOWN SSA VALUE>>, <<UNKNOWN SSA VALUE>>, <<UNKNOWN SSA VALUE>>, <<UNKNOWN SSA VALUE>>) ({
^bb0(%arg0: index, %arg1: index):
%1 = "arith.index_cast"(%arg1) : (index) -> i32
%2:2 = "pto.plt_b32"(%1) : (i32) -> (!pto.mask<b32>, i32)
%3 = "arith.index_cast"(%2#1) : (i32) -> index
%4 = "pto.vlds"(<<UNKNOWN SSA VALUE>>, %arg0) : (!pto.ptr<f32, ub>, index) -> !pto.vreg<64xf32>
%5 = "pto.vlds"(<<UNKNOWN SSA VALUE>>, %arg0) : (!pto.ptr<f32, ub>, index) -> !pto.vreg<64xf32>
%6 = "pto.vadd"(%4, %5, %2#0) : (!pto.vreg<64xf32>, !pto.vreg<64xf32>, !pto.mask<b32>) -> !pto.vreg<64xf32>
"pto.vsts"(%6, <<UNKNOWN SSA VALUE>>, %arg0, %2#0) : (!pto.vreg<64xf32>, !pto.ptr<f32, ub>, index, !pto.mask<b32>) -> ()
"scf.yield"(%3) : (index) -> ()
}) : (index, index, index, index) -> index
bug.pto:42:20: note: - use: %50 = "arith.addi"(<<UNKNOWN SSA VALUE>>, %48) <{overflowFlags = #arith.overflow<none>}> : (index, index) -> index
%setup_use = arith.addi %18, %c0_7 : index
^
LLVM ERROR: operation destroyed but still has uses
PLEASE submit a bug report to https://github.com/llvm/llvm-project/issues/ and include the crash backtrace.
Stack dump:
0. Program arguments: /home/cplop/code/pto/PTOAS/build/tools/pto-test-opt/pto-test-opt -pto-low-level-loop-fusion bug.pto
#0 0x00007d403dbeaea2 llvm::sys::PrintStackTrace(llvm::raw_ostream&, int) (/home/cplop/code/pto/llvm-project/build-shared/lib/libLLVMSupport.so.19.1+0x1eaea2)
#1 0x00007d403dbe7f1f llvm::sys::RunSignalHandlers() (/home/cplop/code/pto/llvm-project/build-shared/lib/libLLVMSupport.so.19.1+0x1e7f1f)
#2 0x00007d403dbe8065 SignalHandler(int) Signals.cpp:0:0
#3 0x00007d403d245330 (/lib/x86_64-linux-gnu/libc.so.6+0x45330)
#4 0x00007d403d29ec0c pthread_kill (/lib/x86_64-linux-gnu/libc.so.6+0x9ec0c)
#5 0x00007d403d24527e raise (/lib/x86_64-linux-gnu/libc.so.6+0x4527e)
#6 0x00007d403d2288ff abort (/lib/x86_64-linux-gnu/libc.so.6+0x288ff)
#7 0x00007d403da5631d CompareNumbers(char const*&, char const*&, char const*, char const*, double, double, std::__cxx11::basic_string<char, std::char_traits<char>, std::allocator<char>>*) (.cold) FileUtilities.cpp:0:0
#8 0x00007d403daf9d2e (/home/cplop/code/pto/llvm-project/build-shared/lib/libLLVMSupport.so.19.1+0xf9d2e)
#9 0x00007d403df754ef mlir::Operation::~Operation() (/home/cplop/code/pto/llvm-project/build-shared/lib/libMLIRIR.so.19.1+0x1754ef)
#10 0x00007d403df75829 mlir::Operation::erase() (/home/cplop/code/pto/llvm-project/build-shared/lib/libMLIRIR.so.19.1+0x175829)
#11 0x0000616b70d0b76b mlir::OpState::erase() (/home/cplop/code/pto/PTOAS/build/tools/pto-test-opt/pto-test-opt+0x103776b)
#12 0x0000616b70d11711 (anonymous namespace)::fuseStageRun(llvm::SmallVectorImpl<(anonymous namespace)::StageInfo>&, llvm::raw_ostream*) PTOLowLevelLoopFusion.cpp:0:0
#13 0x0000616b70d1187a (anonymous namespace)::fuseStageRunsInBlock(mlir::Block&, llvm::raw_ostream*) PTOLowLevelLoopFusion.cpp:0:0
#14 0x0000616b70d1194b (anonymous namespace)::PTOLowLevelLoopFusionPass::runOnOperation()::'lambda'(mlir::pto::FusionRegionOp)::operator()(mlir::pto::FusionRegionOp) const PTOLowLevelLoopFusion.cpp:0:0
#15 0x0000616b70d18ecf _ZZN4mlir6detail4walkILNS_9WalkOrderE1ENS_15ForwardIteratorEZN12_GLOBAL__N_125PTOLowLevelLoopFusionPass14runOnOperationEvEUlNS_3pto14FusionRegionOpEE_S7_vEENSt9enable_ifIXaantsrSt11disjunctionIJSt7is_sameIT2_PNS_9OperationEESB_ISC_PNS_6RegionEESB_ISC_PNS_5BlockEEEE5valuesrSB_IT3_vE5valueESN_E4typeESE_OT1_ENKUlSE_E_clESE_ PTOLowLevelLoopFusion.cpp:0:0
#16 0x0000616b70d1be5f _ZN4llvm12function_refIFvPN4mlir9OperationEEE11callback_fnIZNS1_6detail4walkILNS1_9WalkOrderE1ENS1_15ForwardIteratorEZN12_GLOBAL__N_125PTOLowLevelLoopFusionPass14runOnOperationEvEUlNS1_3pto14FusionRegionOpEE_SE_vEENSt9enable_ifIXaantsrSt11disjunctionIJSt7is_sameIT2_S3_ESI_ISJ_PNS1_6RegionEESI_ISJ_PNS1_5BlockEEEE5valuesrSI_IT3_vE5valueESS_E4typeES3_OT1_EUlS3_E_EEvlS3_ PTOLowLevelLoopFusion.cpp:0:0
#17 0x0000616b7062a0eb llvm::function_ref<void (mlir::Operation*)>::operator()(mlir::Operation*) const (/home/cplop/code/pto/PTOAS/build/tools/pto-test-opt/pto-test-opt+0x9560eb)
#18 0x0000616b705b29fe void mlir::detail::walk<mlir::ForwardIterator>(mlir::Operation*, llvm::function_ref<void (mlir::Operation*)>, mlir::WalkOrder) (/home/cplop/code/pto/PTOAS/build/tools/pto-test-opt/pto-test-opt+0x8de9fe)
#19 0x0000616b705b296d void mlir::detail::walk<mlir::ForwardIterator>(mlir::Operation*, llvm::function_ref<void (mlir::Operation*)>, mlir::WalkOrder) (/home/cplop/code/pto/PTOAS/build/tools/pto-test-opt/pto-test-opt+0x8de96d)
#20 0x0000616b70d18f44 _ZN4mlir6detail4walkILNS_9WalkOrderE1ENS_15ForwardIteratorEZN12_GLOBAL__N_125PTOLowLevelLoopFusionPass14runOnOperationEvEUlNS_3pto14FusionRegionOpEE_S7_vEENSt9enable_ifIXaantsrSt11disjunctionIJSt7is_sameIT2_PNS_9OperationEESB_ISC_PNS_6RegionEESB_ISC_PNS_5BlockEEEE5valuesrSB_IT3_vE5valueESN_E4typeESE_OT1_ PTOLowLevelLoopFusion.cpp:0:0
#21 0x0000616b70d16426 _ZN4mlir9Operation4walkILNS_9WalkOrderE1ENS_15ForwardIteratorEZN12_GLOBAL__N_125PTOLowLevelLoopFusionPass14runOnOperationEvEUlNS_3pto14FusionRegionOpEE_vEENSt9enable_ifIXeqsrN4llvm15function_traitsINSt5decayIT1_E4typeEXsrSt8is_classISF_E5valueEEE8num_argsLi1EET2_E4typeEOSD_ PTOLowLevelLoopFusion.cpp:0:0
#22 0x0000616b70d13d29 _ZN4mlir7OpState4walkILNS_9WalkOrderE1ENS_15ForwardIteratorEZN12_GLOBAL__N_125PTOLowLevelLoopFusionPass14runOnOperationEvEUlNS_3pto14FusionRegionOpEE_vEENSt9enable_ifIXeqsrN4llvm15function_traitsINSt5decayIT1_E4typeEXsrSt8is_classISF_E5valueEEE8num_argsLi1EET2_E4typeEOSD_ PTOLowLevelLoopFusion.cpp:0:0
#23 0x0000616b70d11b46 (anonymous namespace)::PTOLowLevelLoopFusionPass::runOnOperation() PTOLowLevelLoopFusion.cpp:0:0
#24 0x00007d403e1bb3ee mlir::detail::OpToOpPassAdaptor::run(mlir::Pass*, mlir::Operation*, mlir::AnalysisManager, bool, unsigned int) (/home/cplop/code/pto/llvm-project/build-shared/lib/libMLIRPass.so.19.1+0x243ee)
#25 0x00007d403e1bba20 mlir::detail::OpToOpPassAdaptor::runPipeline(mlir::OpPassManager&, mlir::Operation*, mlir::AnalysisManager, bool, unsigned int, mlir::PassInstrumentor*, mlir::PassInstrumentation::PipelineParentInfo const*) (/home/cplop/code/pto/llvm-project/build-shared/lib/libMLIRPass.so.19.1+0x24a20)
#26 0x00007d403e1bc915 mlir::PassManager::run(mlir::Operation*) (/home/cplop/code/pto/llvm-project/build-shared/lib/libMLIRPass.so.19.1+0x25915)
#27 0x00007d403e204fbf performActions(llvm::raw_ostream&, std::shared_ptr<llvm::SourceMgr> const&, mlir::MLIRContext*, mlir::MlirOptMainConfig const&) MlirOptMain.cpp:0:0
#28 0x00007d403e205813 processBuffer(llvm::raw_ostream&, std::unique_ptr<llvm::MemoryBuffer, std::default_delete<llvm::MemoryBuffer>>, mlir::MlirOptMainConfig const&, mlir::DialectRegistry&, llvm::ThreadPoolInterface*) MlirOptMain.cpp:0:0
#29 0x00007d403e20595c llvm::LogicalResult llvm::function_ref<llvm::LogicalResult (std::unique_ptr<llvm::MemoryBuffer, std::default_delete<llvm::MemoryBuffer>>, llvm::raw_ostream&)>::callback_fn<mlir::MlirOptMain(llvm::raw_ostream&, std::unique_ptr<llvm::MemoryBuffer, std::default_delete<llvm::MemoryBuffer>>, mlir::DialectRegistry&, mlir::MlirOptMainConfig const&)::'lambda'(std::unique_ptr<llvm::MemoryBuffer, std::default_delete<llvm::MemoryBuffer>>, llvm::raw_ostream&)>(long, std::unique_ptr<llvm::MemoryBuffer, std::default_delete<llvm::MemoryBuffer>>, llvm::raw_ostream&) MlirOptMain.cpp:0:0
#30 0x00007d403dde81fe mlir::splitAndProcessBuffer(std::unique_ptr<llvm::MemoryBuffer, std::default_delete<llvm::MemoryBuffer>>, llvm::function_ref<llvm::LogicalResult (std::unique_ptr<llvm::MemoryBuffer, std::default_delete<llvm::MemoryBuffer>>, llvm::raw_ostream&)>, llvm::raw_ostream&, llvm::StringRef, llvm::StringRef) (/home/cplop/code/pto/llvm-project/build-shared/lib/libMLIRSupport.so.19.1+0x211fe)
#31 0x00007d403e1fcbbb mlir::MlirOptMain(llvm::raw_ostream&, std::unique_ptr<llvm::MemoryBuffer, std::default_delete<llvm::MemoryBuffer>>, mlir::DialectRegistry&, mlir::MlirOptMainConfig const&) (/home/cplop/code/pto/llvm-project/build-shared/lib/libMLIROptLib.so.19.1+0xabbb)
#32 0x00007d403e205aac mlir::MlirOptMain(int, char**, llvm::StringRef, llvm::StringRef, mlir::DialectRegistry&) (/home/cplop/code/pto/llvm-project/build-shared/lib/libMLIROptLib.so.19.1+0x13aac)
#33 0x00007d403e205fbf mlir::MlirOptMain(int, char**, llvm::StringRef, mlir::DialectRegistry&) (/home/cplop/code/pto/llvm-project/build-shared/lib/libMLIROptLib.so.19.1+0x13fbf)
#34 0x0000616b6fd9e67f main (/home/cplop/code/pto/PTOAS/build/tools/pto-test-opt/pto-test-opt+0xca67f)
#35 0x00007d403d22a1ca (/lib/x86_64-linux-gnu/libc.so.6+0x2a1ca)
#36 0x00007d403d22a28b __libc_start_main (/lib/x86_64-linux-gnu/libc.so.6+0x2a28b)
#37 0x0000616b6fd8f545 _start (/home/cplop/code/pto/PTOAS/build/tools/pto-test-opt/pto-test-opt+0xbb545)
Aborted (core dumped)
Git commit
efc4d3e
Host platform
None
Target Ascend arch (if relevant)
None
PTOAS build level (if relevant)
None
Component
Verifier / IR semantics (lib/PTO/IR)
Description
提单:pto_low_level_loop_fusion — interstage setup 引用前序 loop result 后 erase abort
1. 问题来源
pto_low_level_loop_fusion(-pto-low-level-loop-fusion)$PTOAS_ROOT/lib/PTO/Transforms/TileFusion/PTOLowLevelLoopFusion.cppfuseStageRun→setupOp->moveBefore(fusedOuterLoop)→stage.getOuterLoop().erase()2. 为什么是 bug
collectStageRunFrom允许 stage 之间的 pure op 作为setupOps(isInterstageSetupOp)。这些 op 可以合法引用前一个
scf.for的 result。融合时:
buildFusedLoopNestAtLevel只把原 loop result 记进IRMapping(mapFusedLoopResults)replaceAllUsesWithsetupOps被moveBefore(fusedOuterLoop),仍指向旧 loop resulterase()旧 loop →LLVM ERROR: operation destroyed but still has uses(abort)3. 实证
bug.pto(interstagearith.addi使用前序 loop result)destroyed but still has usescontrol.pto(同一位置 addi 不依赖 loop result)bash test/pto_low_level_loop_fusion_bug_1/reproduce.sh # 预期: 差分复现成功4. 修复建议
在
erase旧 loop 之前,把每个原scf.for的 result 对所有外部 use(含setupOps)replaceAllUsesWith到 fused loop 对应 result;或拒绝收集「使用前序 stage loop result」的 setup op。Reproduction (minimal)
bug.pto
control.pto
两个文件差异:
reproduce.sh
Expected behavior
编译通过
Actual behavior / error logs
Git commit
efc4d3e
Host platform
None
Target Ascend arch (if relevant)
None
PTOAS build level (if relevant)
None