This patch moves the validation logic of delinearization results from DA to Delinearization. Also call it in `printDelinearization` to test its behavior. The motivation is as follows: - Almost the same code exists in `tryDelinearizeFixedSize` and `tryDelinearizeParametricSize`. Consolidating it in Delinearization avoids code duplication. - Currently this validation logic is not well tested. Moving it to Delinearization allows us to write regression tests easily. This patch changes the test outputs and debug messages, but otherwise NFCI.
78 lines
3.3 KiB
LLVM
78 lines
3.3 KiB
LLVM
; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --version 5
|
|
; RUN: opt < %s -passes='print<delinearization>' -disable-output 2>&1 | FileCheck %s
|
|
|
|
target datalayout = "e-m:e-i64:64-f80:128-n8:16:32:64-S128"
|
|
|
|
; Function Attrs: noinline nounwind uwtable
|
|
define void @mat_mul(ptr %C, ptr %A, ptr %B, i64 %N) !kernel_arg_addr_space !2 !kernel_arg_access_qual !3 !kernel_arg_type !4 !kernel_arg_base_type !4 !kernel_arg_type_qual !5 {
|
|
; CHECK-LABEL: 'mat_mul'
|
|
; CHECK-NEXT: Inst: %tmp = load float, ptr %arrayidx, align 4
|
|
; CHECK-NEXT: AccessFunction: {(4 * %N * %call),+,4}<%for.inc>
|
|
; CHECK-NEXT: Base offset: %A
|
|
; CHECK-NEXT: ArrayDecl[UnknownSize][%N] with elements of 4 bytes.
|
|
; CHECK-NEXT: ArrayRef[%call][{0,+,1}<nuw><nsw><%for.inc>]
|
|
; CHECK-NEXT: Delinearization validation: Succeeded
|
|
; CHECK-EMPTY:
|
|
; CHECK-NEXT: Inst: %tmp5 = load float, ptr %arrayidx4, align 4
|
|
; CHECK-NEXT: AccessFunction: {(4 * %call1),+,(4 * %N)}<%for.inc>
|
|
; CHECK-NEXT: Base offset: %B
|
|
; CHECK-NEXT: ArrayDecl[UnknownSize][%N] with elements of 4 bytes.
|
|
; CHECK-NEXT: ArrayRef[{0,+,1}<nuw><nsw><%for.inc>][%call1]
|
|
; CHECK-NEXT: Delinearization validation: Failed
|
|
;
|
|
entry:
|
|
br label %entry.split
|
|
|
|
entry.split: ; preds = %entry
|
|
%call = tail call i64 @_Z13get_global_idj(i32 0)
|
|
%call1 = tail call i64 @_Z13get_global_idj(i32 1)
|
|
%cmp1 = icmp sgt i64 %N, 0
|
|
%mul = mul nsw i64 %call, %N
|
|
br i1 %cmp1, label %for.inc.lr.ph, label %for.end
|
|
|
|
for.inc.lr.ph: ; preds = %entry.split
|
|
br label %for.inc
|
|
|
|
for.inc: ; preds = %for.inc.lr.ph, %for.inc
|
|
%acc.03 = phi float [ 0.000000e+00, %for.inc.lr.ph ], [ %tmp6, %for.inc ]
|
|
%m.02 = phi i64 [ 0, %for.inc.lr.ph ], [ %inc, %for.inc ]
|
|
%add = add nsw i64 %m.02, %mul
|
|
%arrayidx = getelementptr inbounds float, ptr %A, i64 %add
|
|
%tmp = load float, ptr %arrayidx, align 4
|
|
%mul2 = mul nsw i64 %m.02, %N
|
|
%add3 = add nsw i64 %mul2, %call1
|
|
%arrayidx4 = getelementptr inbounds float, ptr %B, i64 %add3
|
|
%tmp5 = load float, ptr %arrayidx4, align 4
|
|
%tmp6 = tail call float @llvm.fmuladd.f32(float %tmp, float %tmp5, float %acc.03)
|
|
%inc = add nuw nsw i64 %m.02, 1
|
|
%exitcond = icmp ne i64 %inc, %N
|
|
br i1 %exitcond, label %for.inc, label %for.cond.for.end_crit_edge
|
|
|
|
for.cond.for.end_crit_edge: ; preds = %for.inc
|
|
%.lcssa = phi float [ %tmp6, %for.inc ]
|
|
br label %for.end
|
|
|
|
for.end: ; preds = %for.cond.for.end_crit_edge, %entry.split
|
|
%acc.0.lcssa = phi float [ %.lcssa, %for.cond.for.end_crit_edge ], [ 0.000000e+00, %entry.split ]
|
|
%add7 = add nsw i64 %mul, %call1
|
|
%arrayidx8 = getelementptr inbounds float, ptr %C, i64 %add7
|
|
store float %acc.0.lcssa, ptr %arrayidx8, align 4
|
|
ret void
|
|
}
|
|
|
|
; Function Attrs: nounwind readnone
|
|
declare i64 @_Z13get_global_idj(i32)
|
|
|
|
; Function Attrs: nounwind readnone speculatable
|
|
declare float @llvm.fmuladd.f32(float, float, float)
|
|
|
|
!llvm.module.flags = !{!0}
|
|
!llvm.ident = !{!1}
|
|
|
|
!0 = !{i32 1, !"wchar_size", i32 4}
|
|
!1 = !{!"clang version 5.0.0 (trunk 303846) (llvm/trunk 303834)"}
|
|
!2 = !{i32 1, i32 1, i32 1, i32 0}
|
|
!3 = !{!"none", !"none", !"none", !"none"}
|
|
!4 = !{!"float*", !"float*", !"float*", !"long"}
|
|
!5 = !{!"", !"", !"", !""}
|