[LLVMGPU] Add loop invariant code motion before software pipelining (#12540)
This cleans up the IR and prevents making invariant op loop carried
dependency.
diff --git a/compiler/src/iree/compiler/Codegen/LLVMGPU/Passes.cpp b/compiler/src/iree/compiler/Codegen/LLVMGPU/Passes.cpp
index 44ec121..30186c6 100644
--- a/compiler/src/iree/compiler/Codegen/LLVMGPU/Passes.cpp
+++ b/compiler/src/iree/compiler/Codegen/LLVMGPU/Passes.cpp
@@ -215,6 +215,9 @@
nestedModulePM.addNestedPass<func::FuncOp>(
createOptimizeVectorTransferPass());
+ // Hoist loop invariant code to avoid pipelining it.
+ nestedModulePM.addNestedPass<func::FuncOp>(
+ createLoopInvariantCodeMotionPass());
// Pipeline memory operations.
nestedModulePM.addNestedPass<func::FuncOp>(createGPUPipeliningPass());
}
@@ -270,6 +273,9 @@
nestedModulePM.addPass(createCanonicalizerPass());
nestedModulePM.addPass(createCSEPass());
+ // Hoist loop invariant code to avoid pipelining it.
+ nestedModulePM.addNestedPass<func::FuncOp>(
+ createLoopInvariantCodeMotionPass());
PipeliningSchedulingStrategy schedule =
llvmgpuUseMMASync ? PipeliningSchedulingStrategy::nvidiaTensorCore
: PipeliningSchedulingStrategy::loadGlobalStage0;