[llvm] [LV] Handle complex multiply reductions (PR #204349)

Ashutosh Nema via llvm-commits llvm-commits at lists.llvm.org
Thu Sep 3 00:50:42 PDT 2026


================
@@ -0,0 +1,165 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt < %s -passes=loop-vectorize -force-vector-interleave=1 -force-vector-width=4 -S | FileCheck %s
+
+define void @reduction_complex_prod_(i64 %n, ptr %A, ptr %R) {
+; CHECK-LABEL: define void @reduction_complex_prod_(
+; CHECK-SAME: i64 [[N:%.*]], ptr [[A:%.*]], ptr [[R:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[R_GEP_IM:%.*]] = getelementptr inbounds nuw i8, ptr [[R]], i64 4
+; CHECK-NEXT:    [[R_RE:%.*]] = load float, ptr [[R]], align 4
+; CHECK-NEXT:    [[R_IM:%.*]] = load float, ptr [[R_GEP_IM]], align 4
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], 4
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <4 x float> zeroinitializer, float [[R_IM]], i32 0
+; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <4 x float> splat (float 1.000000e+00), float [[R_RE]], i32 0
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_PHI:%.*]] = phi <4 x float> [ [[TMP0]], %[[VECTOR_PH]] ], [ [[TMP34:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_PHI1:%.*]] = phi <4 x float> [ [[TMP1]], %[[VECTOR_PH]] ], [ [[TMP31:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = add i64 [[INDEX]], 1
+; CHECK-NEXT:    [[TMP3:%.*]] = add i64 [[INDEX]], 2
+; CHECK-NEXT:    [[TMP4:%.*]] = add i64 [[INDEX]], 3
+; CHECK-NEXT:    [[TMP5:%.*]] = getelementptr [8 x i8], ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP6:%.*]] = getelementptr [8 x i8], ptr [[A]], i64 [[TMP2]]
+; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr [8 x i8], ptr [[A]], i64 [[TMP3]]
+; CHECK-NEXT:    [[TMP8:%.*]] = getelementptr [8 x i8], ptr [[A]], i64 [[TMP4]]
+; CHECK-NEXT:    [[TMP9:%.*]] = load float, ptr [[TMP5]], align 4
+; CHECK-NEXT:    [[TMP10:%.*]] = load float, ptr [[TMP6]], align 4
+; CHECK-NEXT:    [[TMP11:%.*]] = load float, ptr [[TMP7]], align 4
+; CHECK-NEXT:    [[TMP12:%.*]] = load float, ptr [[TMP8]], align 4
+; CHECK-NEXT:    [[TMP13:%.*]] = insertelement <4 x float> poison, float [[TMP9]], i32 0
+; CHECK-NEXT:    [[TMP14:%.*]] = insertelement <4 x float> [[TMP13]], float [[TMP10]], i32 1
+; CHECK-NEXT:    [[TMP15:%.*]] = insertelement <4 x float> [[TMP14]], float [[TMP11]], i32 2
+; CHECK-NEXT:    [[TMP16:%.*]] = insertelement <4 x float> [[TMP15]], float [[TMP12]], i32 3
+; CHECK-NEXT:    [[TMP17:%.*]] = getelementptr i8, ptr [[TMP5]], i64 4
+; CHECK-NEXT:    [[TMP18:%.*]] = getelementptr i8, ptr [[TMP6]], i64 4
+; CHECK-NEXT:    [[TMP19:%.*]] = getelementptr i8, ptr [[TMP7]], i64 4
+; CHECK-NEXT:    [[TMP20:%.*]] = getelementptr i8, ptr [[TMP8]], i64 4
+; CHECK-NEXT:    [[TMP21:%.*]] = load float, ptr [[TMP17]], align 4
+; CHECK-NEXT:    [[TMP22:%.*]] = load float, ptr [[TMP18]], align 4
+; CHECK-NEXT:    [[TMP23:%.*]] = load float, ptr [[TMP19]], align 4
+; CHECK-NEXT:    [[TMP24:%.*]] = load float, ptr [[TMP20]], align 4
+; CHECK-NEXT:    [[TMP25:%.*]] = insertelement <4 x float> poison, float [[TMP21]], i32 0
+; CHECK-NEXT:    [[TMP26:%.*]] = insertelement <4 x float> [[TMP25]], float [[TMP22]], i32 1
+; CHECK-NEXT:    [[TMP27:%.*]] = insertelement <4 x float> [[TMP26]], float [[TMP23]], i32 2
+; CHECK-NEXT:    [[TMP28:%.*]] = insertelement <4 x float> [[TMP27]], float [[TMP24]], i32 3
+; CHECK-NEXT:    [[TMP29:%.*]] = fmul fast <4 x float> [[TMP16]], [[VEC_PHI1]]
+; CHECK-NEXT:    [[TMP30:%.*]] = fmul fast <4 x float> [[TMP28]], [[VEC_PHI]]
+; CHECK-NEXT:    [[TMP31]] = fsub fast <4 x float> [[TMP29]], [[TMP30]]
+; CHECK-NEXT:    [[TMP32:%.*]] = fmul fast <4 x float> [[TMP16]], [[VEC_PHI]]
+; CHECK-NEXT:    [[TMP33:%.*]] = fmul fast <4 x float> [[TMP28]], [[VEC_PHI1]]
+; CHECK-NEXT:    [[TMP34]] = fadd fast <4 x float> [[TMP33]], [[TMP32]]
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP35:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP35]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[LO_RE:%.*]] = shufflevector <4 x float> [[TMP31]], <4 x float> poison, <2 x i32> <i32 0, i32 1>
+; CHECK-NEXT:    [[HI_RE:%.*]] = shufflevector <4 x float> [[TMP31]], <4 x float> poison, <2 x i32> <i32 2, i32 3>
+; CHECK-NEXT:    [[LO_IM:%.*]] = shufflevector <4 x float> [[TMP34]], <4 x float> poison, <2 x i32> <i32 0, i32 1>
+; CHECK-NEXT:    [[HI_IM:%.*]] = shufflevector <4 x float> [[TMP34]], <4 x float> poison, <2 x i32> <i32 2, i32 3>
+; CHECK-NEXT:    [[TMP36:%.*]] = fmul fast <2 x float> [[LO_RE]], [[HI_RE]]
+; CHECK-NEXT:    [[TMP37:%.*]] = fmul fast <2 x float> [[LO_IM]], [[HI_IM]]
+; CHECK-NEXT:    [[TMP38:%.*]] = fmul fast <2 x float> [[LO_RE]], [[HI_IM]]
+; CHECK-NEXT:    [[TMP39:%.*]] = fmul fast <2 x float> [[LO_IM]], [[HI_RE]]
+; CHECK-NEXT:    [[RED_RE:%.*]] = fsub fast <2 x float> [[TMP36]], [[TMP37]]
+; CHECK-NEXT:    [[RED_IM:%.*]] = fadd fast <2 x float> [[TMP38]], [[TMP39]]
+; CHECK-NEXT:    [[LO_RE2:%.*]] = shufflevector <2 x float> [[RED_RE]], <2 x float> poison, <1 x i32> zeroinitializer
+; CHECK-NEXT:    [[HI_RE3:%.*]] = shufflevector <2 x float> [[RED_RE]], <2 x float> poison, <1 x i32> <i32 1>
+; CHECK-NEXT:    [[LO_IM4:%.*]] = shufflevector <2 x float> [[RED_IM]], <2 x float> poison, <1 x i32> zeroinitializer
+; CHECK-NEXT:    [[HI_IM5:%.*]] = shufflevector <2 x float> [[RED_IM]], <2 x float> poison, <1 x i32> <i32 1>
+; CHECK-NEXT:    [[TMP40:%.*]] = fmul fast <1 x float> [[LO_RE2]], [[HI_RE3]]
+; CHECK-NEXT:    [[TMP41:%.*]] = fmul fast <1 x float> [[LO_IM4]], [[HI_IM5]]
+; CHECK-NEXT:    [[TMP42:%.*]] = fmul fast <1 x float> [[LO_RE2]], [[HI_IM5]]
+; CHECK-NEXT:    [[TMP43:%.*]] = fmul fast <1 x float> [[LO_IM4]], [[HI_RE3]]
+; CHECK-NEXT:    [[RED_RE6:%.*]] = fsub fast <1 x float> [[TMP40]], [[TMP41]]
+; CHECK-NEXT:    [[RED_IM7:%.*]] = fadd fast <1 x float> [[TMP42]], [[TMP43]]
+; CHECK-NEXT:    [[FINAL_RE:%.*]] = extractelement <1 x float> [[RED_RE6]], i64 0
+; CHECK-NEXT:    [[FINAL_IM:%.*]] = extractelement <1 x float> [[RED_IM7]], i64 0
----------------
nema-ashutosh wrote:

It looks like we emit the same reduction tree twice here, once for each part ?

https://github.com/llvm/llvm-project/pull/204349


More information about the llvm-commits mailing list