Author: Sander de Smalen Date: 2026-10-07T09:32:48Z New Revision: 708fee365beae021be8378300a4fbe6e2d9508d4
URL: https://github.com/llvm/llvm-project/commit/708fee365beae021be8378300a4fbe6e2d9508d4 DIFF: https://github.com/llvm/llvm-project/commit/708fee365beae021be8378300a4fbe6e2d9508d4.diff LOG: [AArch64][SME] Allow more inlining when SME attributes are incompatible. (#223393) At the moment, 'areInlineCompatible' is very strict as it conservatively disallows inlining any callee if they use intrinsics and have incompatible SME attributes. This PR relaxes those constraints by allowing more intrinsics. It also updates the Clang diagnostic to match the 'new' behaviour that a function is no longer inlined despite 'always_inline' when they are not inline compatible. This is an alternative approach to #218727. This PR holds on to the approach of not inlining unless proven safe to do so, as opposed to relying on the user to know what they're doing (#218727). This a first step in trying to fix https://github.com/llvm/llvm-project/issues/217639 in a better way. Added: llvm/test/Transforms/Inline/AArch64/sme-pstatesm-inline-transitive.ll Modified: clang/include/clang/Basic/DiagnosticFrontendKinds.td clang/test/CodeGen/AArch64/sme-inline-streaming-attrs.c llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp llvm/test/Transforms/Inline/AArch64/sme-pstatesm-attrs.ll Removed: llvm/test/Transforms/Inline/AArch64/sme-pstatesm-attrs-low-threshold.ll ################################################################################ diff --git a/clang/include/clang/Basic/DiagnosticFrontendKinds.td b/clang/include/clang/Basic/DiagnosticFrontendKinds.td index 5a9b79f0659bd1..232475b6fb5a77 100644 --- a/clang/include/clang/Basic/DiagnosticFrontendKinds.td +++ b/clang/include/clang/Basic/DiagnosticFrontendKinds.td @@ -324,7 +324,7 @@ def err_function_always_inline_attribute_mismatch : Error< "always_inline function %1 and its caller %0 have mismatching %2 attributes">; def warn_function_always_inline_attribute_mismatch : Warning< "always_inline function %1 and its caller %0 have mismatching %2 attributes, " - "inlining may change runtime behaviour">, InGroup<AArch64SMEAttributes>; + "%1 may not be inlined">, InGroup<AArch64SMEAttributes>; def err_function_always_inline_new_za : Error< "always_inline function %0 has new za state">; def err_function_always_inline_new_zt0 diff --git a/clang/test/CodeGen/AArch64/sme-inline-streaming-attrs.c b/clang/test/CodeGen/AArch64/sme-inline-streaming-attrs.c index 68102c9ded40c4..ba9f24032d2745 100644 --- a/clang/test/CodeGen/AArch64/sme-inline-streaming-attrs.c +++ b/clang/test/CodeGen/AArch64/sme-inline-streaming-attrs.c @@ -26,7 +26,7 @@ void caller(void) { #ifdef TEST_COMPATIBLE void caller_compatible(void) __arm_streaming_compatible { - inlined_fn(); // expected-warning {{always_inline function 'inlined_fn' and its caller 'caller_compatible' have mismatching streaming attributes, inlining may change runtime behaviour}} + inlined_fn(); // expected-warning {{always_inline function 'inlined_fn' and its caller 'caller_compatible' have mismatching streaming attributes, 'inlined_fn' may not be inlined}} inlined_fn_streaming_compatible(); inlined_fn_streaming(); // expected-error {{always_inline function 'inlined_fn_streaming' and its caller 'caller_compatible' have mismatching streaming attributes}} inlined_fn_local(); // expected-error {{always_inline function 'inlined_fn_local' and its caller 'caller_compatible' have mismatching streaming attributes}} @@ -35,7 +35,7 @@ void caller_compatible(void) __arm_streaming_compatible { #ifdef TEST_STREAMING void caller_streaming(void) __arm_streaming { - inlined_fn(); // expected-warning {{always_inline function 'inlined_fn' and its caller 'caller_streaming' have mismatching streaming attributes, inlining may change runtime behaviour}} + inlined_fn(); // expected-warning {{always_inline function 'inlined_fn' and its caller 'caller_streaming' have mismatching streaming attributes, 'inlined_fn' may not be inlined}} inlined_fn_streaming_compatible(); inlined_fn_streaming(); inlined_fn_local(); @@ -45,7 +45,7 @@ void caller_streaming(void) __arm_streaming { #ifdef TEST_LOCALLY __arm_locally_streaming void caller_local(void) { - inlined_fn(); // expected-warning {{always_inline function 'inlined_fn' and its caller 'caller_local' have mismatching streaming attributes, inlining may change runtime behaviour}} + inlined_fn(); // expected-warning {{always_inline function 'inlined_fn' and its caller 'caller_local' have mismatching streaming attributes, 'inlined_fn' may not be inlined}} inlined_fn_streaming_compatible(); inlined_fn_streaming(); inlined_fn_local(); diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp index 562a8588e33fdb..986c21c50cad5e 100644 --- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp +++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp @@ -11,6 +11,7 @@ #include "AArch64PerfectShuffle.h" #include "AArch64SMEAttributes.h" #include "MCTargetDesc/AArch64AddressingModes.h" +#include "llvm/ADT/BitmaskEnum.h" #include "llvm/ADT/DenseMap.h" #include "llvm/ADT/bit.h" #include "llvm/Analysis/LoopInfo.h" @@ -230,30 +231,126 @@ static cl::opt<bool> EnableFixedwidthAutovecInStreamingMode( static cl::opt<bool> EnableScalableAutovecInStreamingMode( "enable-scalable-autovec-in-streaming-mode", cl::init(false), cl::Hidden); -static bool isSMEABIRoutineCall(const CallInst &CI, - const AArch64TargetLowering &TLI) { - const auto *F = CI.getCalledFunction(); - return F && - SMEAttrs(F->getName(), TLI.getRuntimeLibcallsInfo()).isSMEABIRoutine(); +// Enum to record why an operation is incompatible in a diff erent streaming +// mode. +enum class StreamingIncompatibility : uint8_t { + None = 0, + VScaleDependent = 1 << 0, + HasIncompatibleInstruction = 1 << 1, + FullyIncompatible = VScaleDependent | HasIncompatibleInstruction, + LLVM_MARK_AS_BITMASK_ENUM(/*LargestFlag=*/HasIncompatibleInstruction) +}; + +/// Returns an enum describing whether \p I is an instruction that is +/// incompatible in the mode of the caller. +static StreamingIncompatibility +getCompatibilityForChangeToStreamingMode(const Instruction *I) { + if (auto *II = dyn_cast<IntrinsicInst>(I)) { + unsigned IID = II->getIntrinsicID(); + switch (IID) { + default: + // Conservatively disallow target intrinsics in other modes. + if (Intrinsic::isTargetIntrinsic(IID)) + return StreamingIncompatibility::FullyIncompatible; + break; + case Intrinsic::vscale: + return StreamingIncompatibility::VScaleDependent; + case Intrinsic::masked_gather: + // Instructions that are not available in streaming mode. + // Fixed-length operations can still be code-generated by scalarizing. + // FIXME: Consider FEAT_FA64. + if (I->getOperand(0)->getType()->isScalableTy()) + return StreamingIncompatibility::FullyIncompatible; + break; + case Intrinsic::masked_scatter: + case Intrinsic::masked_compressstore: + case Intrinsic::experimental_vector_histogram_add: + // Instructions that are not available in streaming mode. + // Fixed-length operations can still be code-generated by scalarizing. + // FIXME: Consider FEAT_FA64. + if (I->getOperand(0)->getType()->isScalableTy()) + return StreamingIncompatibility::FullyIncompatible; + break; + } + } + + // Inlining operations on scalable vectors is rejected when + // vscale-dependent operations cannot be safely inlined. + if (I->getType()->isScalableTy() || + any_of(I->operand_values(), + [](const Value *V) { return V->getType()->isScalableTy(); }) || + (isa<GetElementPtrInst>(I) && + cast<GetElementPtrInst>(I)->getSourceElementType()->isScalableTy()) || + (isa<AllocaInst>(I) && cast<AllocaInst>(I)->isScalable())) + return StreamingIncompatibility::VScaleDependent; + + return StreamingIncompatibility::None; } /// Returns true if the function has explicit operations that can only be /// lowered using incompatible instructions for the selected mode. This also /// returns true if the function F may use or modify ZA state. static bool hasPossibleIncompatibleOps(const Function *F, - const AArch64TargetLowering &TLI) { + const AArch64TargetLowering &TLI, + bool ConsiderZA, bool ConsiderZT, + bool ConsiderSM) { + if (!ConsiderZA && !ConsiderZT && !ConsiderSM) + return false; + + bool IsAlwaysInline = F->hasFnAttribute(Attribute::AlwaysInline); + // If the interface has vscale-dependent arguments/return value, then the ACLE + // describes that the function can only have defined behaviour if vscale is + // the same for both streaming/non-streaming mode, meaning that we can assume + // vscale is equivalent. + // FIXME: Handle the case where vscale_range is known and equivalent for + // caller and callee. + bool AssumeVScaleIsEquivalent = + F->getReturnType()->isScalableTy() || + any_of(F->getFunctionType()->params(), + [](const Type *T) { return T->isScalableTy(); }); + for (const BasicBlock &BB : *F) { for (const Instruction &I : BB) { - // Be conservative for now and assume that any call to inline asm or to - // intrinsics could could result in non-streaming ops (e.g. calls to - // @llvm.aarch64.* or @llvm.gather/scatter intrinsics). We can assume that - // all native LLVM instructions can be lowered to compatible instructions. - if (isa<CallInst>(I) && !I.isDebugOrPseudoInst() && - (cast<CallInst>(I).isInlineAsm() || isa<IntrinsicInst>(I) || - isSMEABIRoutineCall(cast<CallInst>(I), TLI))) + if (auto *CB = dyn_cast<CallBase>(&I)) { + SMECallAttrs CallAttrs(*CB, &TLI.getRuntimeLibcallsInfo()); + + // Inline asm must be rejected as it could use SME state. + if (CB->isInlineAsm()) + return true; + + if (CallAttrs.callee().isSMEABIRoutine()) + return true; + + // If the callee has calls to streaming compatible functions, then those + // may have statements which depend on SME state (e.g. vscale or + // explicit reads of PSTATE.SM). If we were to inline those calls, + // behaviour may change after inlining because the streaming-compatible + // function would be executed in a diff erent streaming mode. + // Conservatively avoid inlining for now. + if (ConsiderSM && CallAttrs.callee().hasStreamingCompatibleInterface()) + return true; + } + + if (!ConsiderSM) + continue; + + // Inlining operations on fixed-length vectors when the streaming + // mode does not match is rejected because performance may be impacted. + // This decision should eventually be moved to the cost-model. + if (!IsAlwaysInline && (isa<FixedVectorType>(I.getType()) || + any_of(I.operand_values(), [](const Value *V) { + return isa<FixedVectorType>(V->getType()); + }))) + return true; + + StreamingIncompatibility C = getCompatibilityForChangeToStreamingMode(&I); + if (any(C & StreamingIncompatibility::HasIncompatibleInstruction) || + (!AssumeVScaleIsEquivalent && + any(C & StreamingIncompatibility::VScaleDependent))) return true; } } + return false; } @@ -281,6 +378,48 @@ bool AArch64TTIImpl::isMultiversionedFunction(const Function &F) const { return F.hasFnAttribute("fmv-features"); } +/// The compiler must not inline when that may alter the behavior of the +/// program. This is especially relevant around SME which implements diff erent +/// runtime modes and maintains external state through attributes. +/// +/// The compiler must not inline when: +// +/// * The module is compiled with 'strict-fp' and the called function +/// implements a diff erent FP environment than the caller. +/// +/// * The called function has operations that are incompatible in the mode +/// of the caller, e.g.: +/// * inlining non-streaming-only instructions into a streaming function. +/// * inlining streaming-only instructions into a non-streaming function. +/// +/// Inline asm blocks must be entered and exited in the [streaming] mode of +/// the parent function. There is no language-level mechanism to inform the +/// compiler that a particular inline asm block is streaming compatible, so +/// the compiler must reject this as a candidate for inlining. +/// +/// * The called function contains vscale-dependent operations but otherwise +/// does not take/return VL-dependent arguments (see definition in the +/// ACLE (Arm C/C++ Language Extensions)). +/// +/// * The called function sets up new ZA/ZT state into a function that already +/// has ZA or ZT state, as that is not valid as per the ACLE. +/// +/// The compiler should not inline when: +// +/// * The called function has fixed-length vectors and the caller is in +/// streaming mode, as this may cause performance regressions. This should +/// never result in an error to be reported. +/// +/// * The called function sets up new ZA/ZT state into a function that has no +/// ZA and no ZT state, as the compiler currently cannot transfer the +/// attribute to the caller. It may also impact performance, but that should +/// be covered by AArch64TTIImpl::getInlineCallPenalty(). +/// +/// If a callee is incompatible with its caller, then the inliner should prevent +/// inlining. LLVM's `alwaysinline` attribute has no impact on whether a callee +/// is inline compatible. This is especially relevant with streaming-mode +/// attributes, where `alwaysinline`ing a function into a function with a +/// diff erent streaming mode may otherwise impact runtime behaviour. bool AArch64TTIImpl::areInlineCompatible(const Function *Caller, const Function *Callee) const { SMECallAttrs CallAttrs(*Caller, *Callee); @@ -292,6 +431,9 @@ bool AArch64TTIImpl::areInlineCompatible(const Function *Caller, CallAttrs.callee().hasStreamingInterfaceOrBody()) return false; + if (CallAttrs.callee().isNewZA() || CallAttrs.callee().isNewZT0()) + return false; + // When inlining, we should consider the body of the function, not the // interface. if (CallAttrs.callee().hasStreamingBody()) { @@ -299,15 +441,20 @@ bool AArch64TTIImpl::areInlineCompatible(const Function *Caller, CallAttrs.callee().set(SMEAttrs::SM_Enabled, true); } - if (CallAttrs.callee().isNewZA() || CallAttrs.callee().isNewZT0()) + bool ConsiderZA = CallAttrs.requiresZASave(); + bool ConsiderZT = CallAttrs.requiresPreservingZT0() || + CallAttrs.requiresPreservingAllZAState(); + bool ConsiderSM = CallAttrs.requiresSMChange(); + + // FP environment is interpreted diff erently between streaming mode and + // non-streaming mode, so are not inline-compatible. + // FIXME: Analyze whether the callee actually has any FP operations. + if (ConsiderSM && (Caller->isStrictFP() || Callee->isStrictFP())) return false; - if (CallAttrs.requiresLazySave() || CallAttrs.requiresSMChange() || - CallAttrs.requiresPreservingZT0() || - CallAttrs.requiresPreservingAllZAState()) { - if (hasPossibleIncompatibleOps(Callee, *getTLI())) - return false; - } + if (hasPossibleIncompatibleOps(Callee, *getTLI(), ConsiderZA, ConsiderZT, + ConsiderSM)) + return false; return BaseT::areInlineCompatible(Caller, Callee); } diff --git a/llvm/test/Transforms/Inline/AArch64/sme-pstatesm-attrs-low-threshold.ll b/llvm/test/Transforms/Inline/AArch64/sme-pstatesm-attrs-low-threshold.ll deleted file mode 100644 index 8a608a1b8e156c..00000000000000 --- a/llvm/test/Transforms/Inline/AArch64/sme-pstatesm-attrs-low-threshold.ll +++ /dev/null @@ -1,47 +0,0 @@ -; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 2 -; RUN: opt < %s -mtriple=aarch64-unknown-linux-gnu -mattr=+sme -S -passes=inline -inlinedefault-threshold=1 | FileCheck %s - -; This test sets the inline-threshold to 1 such that by default the call to @streaming_callee is not inlined. -; However, if the call to @streaming_callee requires a streaming-mode change, it should always inline the call because the streaming-mode change is more expensive. -target triple = "aarch64" - -declare void @streaming_compatible_f() #0 "aarch64_pstate_sm_compatible" - -; Function @non_streaming_callee doesn't contain any operations that may use ZA -; state and therefore can be legally inlined into a normal function. -define void @non_streaming_callee() #0 { -; CHECK-LABEL: define void @non_streaming_callee -; CHECK-SAME: () #[[ATTR1:[0-9]+]] { -; CHECK-NEXT: call void @streaming_compatible_f() -; CHECK-NEXT: call void @streaming_compatible_f() -; CHECK-NEXT: ret void -; - call void @streaming_compatible_f() - call void @streaming_compatible_f() - ret void -} - -; Inline call to @non_streaming_callee to remove a streaming mode change. -define void @streaming_caller_inline() #0 "aarch64_pstate_sm_enabled" { -; CHECK-LABEL: define void @streaming_caller_inline -; CHECK-SAME: () #[[ATTR2:[0-9]+]] { -; CHECK-NEXT: call void @streaming_compatible_f() -; CHECK-NEXT: call void @streaming_compatible_f() -; CHECK-NEXT: ret void -; - call void @non_streaming_callee() - ret void -} - -; Don't inline call to @non_streaming_callee when the inline-threshold is set to 1, because it does not eliminate a streaming-mode change. -define void @non_streaming_caller_dont_inline() #0 { -; CHECK-LABEL: define void @non_streaming_caller_dont_inline -; CHECK-SAME: () #[[ATTR1]] { -; CHECK-NEXT: call void @non_streaming_callee() -; CHECK-NEXT: ret void -; - call void @non_streaming_callee() - ret void -} - -attributes #0 = { "target-features"="+sve,+sme" } diff --git a/llvm/test/Transforms/Inline/AArch64/sme-pstatesm-attrs.ll b/llvm/test/Transforms/Inline/AArch64/sme-pstatesm-attrs.ll index 077a3aa49fb414..45c39056266231 100644 --- a/llvm/test/Transforms/Inline/AArch64/sme-pstatesm-attrs.ll +++ b/llvm/test/Transforms/Inline/AArch64/sme-pstatesm-attrs.ll @@ -1,5 +1,5 @@ ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 2 -; RUN: opt < %s -mtriple=aarch64-unknown-linux-gnu -mattr=+sme -S -passes=inline | FileCheck %s +; RUN: opt < %s -mtriple=aarch64-unknown-linux-gnu -mattr=+sve,+sme -S -passes=inline | FileCheck %s declare i32 @llvm.vscale.i32() @@ -7,7 +7,7 @@ declare i32 @llvm.vscale.i32() ; by the other functions below. If we see the call to one of these functions ; being replaced by 'llvm.vscale()', then we know it has been inlined. -define i32 @normal_callee() #0 { +define i32 @normal_callee() { ; CHECK-LABEL: define i32 @normal_callee ; CHECK-SAME: () #[[ATTR1:[0-9]+]] { ; CHECK-NEXT: entry: @@ -19,7 +19,7 @@ entry: ret i32 %res } -define i32 @streaming_callee() #0 "aarch64_pstate_sm_enabled" { +define i32 @streaming_callee() "aarch64_pstate_sm_enabled" { ; CHECK-LABEL: define i32 @streaming_callee ; CHECK-SAME: () #[[ATTR2:[0-9]+]] { ; CHECK-NEXT: entry: @@ -31,7 +31,7 @@ entry: ret i32 %res } -define i32 @locally_streaming_callee() #0 "aarch64_pstate_sm_body" { +define i32 @locally_streaming_callee() "aarch64_pstate_sm_body" { ; CHECK-LABEL: define i32 @locally_streaming_callee ; CHECK-SAME: () #[[ATTR3:[0-9]+]] { ; CHECK-NEXT: entry: @@ -43,7 +43,7 @@ entry: ret i32 %res } -define i32 @streaming_compatible_callee() #0 "aarch64_pstate_sm_compatible" { +define i32 @streaming_compatible_callee() "aarch64_pstate_sm_compatible" { ; CHECK-LABEL: define i32 @streaming_compatible_callee ; CHECK-SAME: () #[[ATTR4:[0-9]+]] { ; CHECK-NEXT: entry: @@ -55,7 +55,7 @@ entry: ret i32 %res } -define i32 @streaming_compatible_locally_streaming_callee() #0 "aarch64_pstate_sm_compatible" "aarch64_pstate_sm_body" { +define i32 @streaming_compatible_locally_streaming_callee() "aarch64_pstate_sm_compatible" "aarch64_pstate_sm_body" { ; CHECK-LABEL: define i32 @streaming_compatible_locally_streaming_callee ; CHECK-SAME: () #[[ATTR5:[0-9]+]] { ; CHECK-NEXT: entry: @@ -84,7 +84,7 @@ entry: ; [ ] N -> SC ; [ ] N -> N + B ; [ ] N -> SC + B -define i32 @normal_caller_normal_callee_inline() #0 { +define i32 @normal_caller_normal_callee_inline() { ; CHECK-LABEL: define i32 @normal_caller_normal_callee_inline ; CHECK-SAME: () #[[ATTR1]] { ; CHECK-NEXT: entry: @@ -101,7 +101,7 @@ entry: ; [ ] N -> SC ; [ ] N -> N + B ; [ ] N -> SC + B -define i32 @normal_caller_streaming_callee_dont_inline() #0 { +define i32 @normal_caller_streaming_callee_dont_inline() { ; CHECK-LABEL: define i32 @normal_caller_streaming_callee_dont_inline ; CHECK-SAME: () #[[ATTR1]] { ; CHECK-NEXT: entry: @@ -118,7 +118,7 @@ entry: ; [x] N -> SC ; [ ] N -> N + B ; [ ] N -> SC + B -define i32 @normal_caller_streaming_compatible_callee_inline() #0 { +define i32 @normal_caller_streaming_compatible_callee_inline() { ; CHECK-LABEL: define i32 @normal_caller_streaming_compatible_callee_inline ; CHECK-SAME: () #[[ATTR1]] { ; CHECK-NEXT: entry: @@ -135,7 +135,7 @@ entry: ; [ ] N -> SC ; [x] N -> N + B ; [ ] N -> SC + B -define i32 @normal_caller_locally_streaming_callee_dont_inline() #0 { +define i32 @normal_caller_locally_streaming_callee_dont_inline() { ; CHECK-LABEL: define i32 @normal_caller_locally_streaming_callee_dont_inline ; CHECK-SAME: () #[[ATTR1]] { ; CHECK-NEXT: entry: @@ -152,7 +152,7 @@ entry: ; [ ] N -> SC ; [ ] N -> N + B ; [x] N -> SC + B -define i32 @normal_caller_streaming_compatible_locally_streaming_callee_dont_inline() #0 { +define i32 @normal_caller_streaming_compatible_locally_streaming_callee_dont_inline() { ; CHECK-LABEL: define i32 @normal_caller_streaming_compatible_locally_streaming_callee_dont_inline ; CHECK-SAME: () #[[ATTR1]] { ; CHECK-NEXT: entry: @@ -169,7 +169,7 @@ entry: ; [ ] S -> SC ; [ ] S -> N + B ; [ ] S -> SC + B -define i32 @streaming_caller_normal_callee_dont_inline() #0 "aarch64_pstate_sm_enabled" { +define i32 @streaming_caller_normal_callee_dont_inline() "aarch64_pstate_sm_enabled" { ; CHECK-LABEL: define i32 @streaming_caller_normal_callee_dont_inline ; CHECK-SAME: () #[[ATTR2]] { ; CHECK-NEXT: entry: @@ -186,7 +186,7 @@ entry: ; [ ] S -> SC ; [ ] S -> N + B ; [ ] S -> SC + B -define i32 @streaming_caller_streaming_callee_inline() #0 "aarch64_pstate_sm_enabled" { +define i32 @streaming_caller_streaming_callee_inline() "aarch64_pstate_sm_enabled" { ; CHECK-LABEL: define i32 @streaming_caller_streaming_callee_inline ; CHECK-SAME: () #[[ATTR2]] { ; CHECK-NEXT: entry: @@ -203,7 +203,7 @@ entry: ; [x] S -> SC ; [ ] S -> N + B ; [ ] S -> SC + B -define i32 @streaming_caller_streaming_compatible_callee_inline() #0 "aarch64_pstate_sm_enabled" { +define i32 @streaming_caller_streaming_compatible_callee_inline() "aarch64_pstate_sm_enabled" { ; CHECK-LABEL: define i32 @streaming_caller_streaming_compatible_callee_inline ; CHECK-SAME: () #[[ATTR2]] { ; CHECK-NEXT: entry: @@ -220,7 +220,7 @@ entry: ; [ ] S -> SC ; [x] S -> N + B ; [ ] S -> SC + B -define i32 @streaming_caller_locally_streaming_callee_inline() #0 "aarch64_pstate_sm_enabled" { +define i32 @streaming_caller_locally_streaming_callee_inline() "aarch64_pstate_sm_enabled" { ; CHECK-LABEL: define i32 @streaming_caller_locally_streaming_callee_inline ; CHECK-SAME: () #[[ATTR2]] { ; CHECK-NEXT: entry: @@ -237,7 +237,7 @@ entry: ; [ ] S -> SC ; [ ] S -> N + B ; [x] S -> SC + B -define i32 @streaming_caller_streaming_compatible_locally_streaming_callee_inline() #0 "aarch64_pstate_sm_enabled" { +define i32 @streaming_caller_streaming_compatible_locally_streaming_callee_inline() "aarch64_pstate_sm_enabled" { ; CHECK-LABEL: define i32 @streaming_caller_streaming_compatible_locally_streaming_callee_inline ; CHECK-SAME: () #[[ATTR2]] { ; CHECK-NEXT: entry: @@ -254,7 +254,7 @@ entry: ; [ ] N + B -> SC ; [ ] N + B -> N + B ; [ ] N + B -> SC + B -define i32 @locally_streaming_caller_normal_callee_dont_inline() #0 "aarch64_pstate_sm_body" { +define i32 @locally_streaming_caller_normal_callee_dont_inline() "aarch64_pstate_sm_body" { ; CHECK-LABEL: define i32 @locally_streaming_caller_normal_callee_dont_inline ; CHECK-SAME: () #[[ATTR3]] { ; CHECK-NEXT: entry: @@ -271,7 +271,7 @@ entry: ; [ ] N + B -> SC ; [ ] N + B -> N + B ; [ ] N + B -> SC + B -define i32 @locally_streaming_caller_streaming_callee_inline() #0 "aarch64_pstate_sm_body" { +define i32 @locally_streaming_caller_streaming_callee_inline() "aarch64_pstate_sm_body" { ; CHECK-LABEL: define i32 @locally_streaming_caller_streaming_callee_inline ; CHECK-SAME: () #[[ATTR3]] { ; CHECK-NEXT: entry: @@ -288,7 +288,7 @@ entry: ; [x] N + B -> SC ; [ ] N + B -> N + B ; [ ] N + B -> SC + B -define i32 @locally_streaming_caller_streaming_compatible_callee_inline() #0 "aarch64_pstate_sm_body" { +define i32 @locally_streaming_caller_streaming_compatible_callee_inline() "aarch64_pstate_sm_body" { ; CHECK-LABEL: define i32 @locally_streaming_caller_streaming_compatible_callee_inline ; CHECK-SAME: () #[[ATTR3]] { ; CHECK-NEXT: entry: @@ -305,7 +305,7 @@ entry: ; [ ] N + B -> SC ; [x] N + B -> N + B ; [ ] N + B -> SC + B -define i32 @locally_streaming_caller_locally_streaming_callee_inline() #0 "aarch64_pstate_sm_body" { +define i32 @locally_streaming_caller_locally_streaming_callee_inline() "aarch64_pstate_sm_body" { ; CHECK-LABEL: define i32 @locally_streaming_caller_locally_streaming_callee_inline ; CHECK-SAME: () #[[ATTR3]] { ; CHECK-NEXT: entry: @@ -322,7 +322,7 @@ entry: ; [ ] N + B -> SC ; [ ] N + B -> N + B ; [x] N + B -> SC + B -define i32 @locally_streaming_caller_streaming_compatible_locally_streaming_callee_inline() #0 "aarch64_pstate_sm_body" { +define i32 @locally_streaming_caller_streaming_compatible_locally_streaming_callee_inline() "aarch64_pstate_sm_body" { ; CHECK-LABEL: define i32 @locally_streaming_caller_streaming_compatible_locally_streaming_callee_inline ; CHECK-SAME: () #[[ATTR3]] { ; CHECK-NEXT: entry: @@ -339,7 +339,7 @@ entry: ; [ ] SC -> SC ; [ ] SC -> N + B ; [ ] SC -> SC + B -define i32 @streaming_compatible_caller_normal_callee_dont_inline() #0 "aarch64_pstate_sm_compatible" { +define i32 @streaming_compatible_caller_normal_callee_dont_inline() "aarch64_pstate_sm_compatible" { ; CHECK-LABEL: define i32 @streaming_compatible_caller_normal_callee_dont_inline ; CHECK-SAME: () #[[ATTR4]] { ; CHECK-NEXT: entry: @@ -356,7 +356,7 @@ entry: ; [ ] SC -> SC ; [ ] SC -> N + B ; [ ] SC -> SC + B -define i32 @streaming_compatible_caller_streaming_callee_dont_inline() #0 "aarch64_pstate_sm_compatible" { +define i32 @streaming_compatible_caller_streaming_callee_dont_inline() "aarch64_pstate_sm_compatible" { ; CHECK-LABEL: define i32 @streaming_compatible_caller_streaming_callee_dont_inline ; CHECK-SAME: () #[[ATTR4]] { ; CHECK-NEXT: entry: @@ -373,7 +373,7 @@ entry: ; [x] SC -> SC ; [ ] SC -> N + B ; [ ] SC -> SC + B -define i32 @streaming_compatible_caller_streaming_compatible_callee_inline() #0 "aarch64_pstate_sm_compatible" { +define i32 @streaming_compatible_caller_streaming_compatible_callee_inline() "aarch64_pstate_sm_compatible" { ; CHECK-LABEL: define i32 @streaming_compatible_caller_streaming_compatible_callee_inline ; CHECK-SAME: () #[[ATTR4]] { ; CHECK-NEXT: entry: @@ -390,7 +390,7 @@ entry: ; [ ] SC -> SC ; [x] SC -> N + B ; [ ] SC -> SC + B -define i32 @streaming_compatible_caller_locally_streaming_callee_dont_inline() #0 "aarch64_pstate_sm_compatible" { +define i32 @streaming_compatible_caller_locally_streaming_callee_dont_inline() "aarch64_pstate_sm_compatible" { ; CHECK-LABEL: define i32 @streaming_compatible_caller_locally_streaming_callee_dont_inline ; CHECK-SAME: () #[[ATTR4]] { ; CHECK-NEXT: entry: @@ -407,7 +407,7 @@ entry: ; [ ] SC -> SC ; [ ] SC -> N + B ; [x] SC -> SC + B -define i32 @streaming_compatible_caller_streaming_compatible_locally_streaming_callee_dont_inline() #0 "aarch64_pstate_sm_compatible" { +define i32 @streaming_compatible_caller_streaming_compatible_locally_streaming_callee_dont_inline() "aarch64_pstate_sm_compatible" { ; CHECK-LABEL: define i32 @streaming_compatible_caller_streaming_compatible_locally_streaming_callee_dont_inline ; CHECK-SAME: () #[[ATTR4]] { ; CHECK-NEXT: entry: @@ -423,7 +423,7 @@ entry: ; [ ] SC + B -> SC ; [ ] SC + B -> N + B ; [ ] SC + B -> SC + B -define i32 @streaming_compatible_locally_streaming_caller_normal_callee_dont_inline() #0 "aarch64_pstate_sm_compatible" "aarch64_pstate_sm_body" { +define i32 @streaming_compatible_locally_streaming_caller_normal_callee_dont_inline() "aarch64_pstate_sm_compatible" "aarch64_pstate_sm_body" { ; CHECK-LABEL: define i32 @streaming_compatible_locally_streaming_caller_normal_callee_dont_inline ; CHECK-SAME: () #[[ATTR5]] { ; CHECK-NEXT: entry: @@ -440,7 +440,7 @@ entry: ; [ ] SC + B -> SC ; [ ] SC + B -> N + B ; [ ] SC + B -> SC + B -define i32 @streaming_compatible_locally_streaming_caller_streaming_callee_inline() #0 "aarch64_pstate_sm_compatible" "aarch64_pstate_sm_body" { +define i32 @streaming_compatible_locally_streaming_caller_streaming_callee_inline() "aarch64_pstate_sm_compatible" "aarch64_pstate_sm_body" { ; CHECK-LABEL: define i32 @streaming_compatible_locally_streaming_caller_streaming_callee_inline ; CHECK-SAME: () #[[ATTR5]] { ; CHECK-NEXT: entry: @@ -457,7 +457,7 @@ entry: ; [x] SC + B -> SC ; [ ] SC + B -> N + B ; [ ] SC + B -> SC + B -define i32 @streaming_compatible_locally_streaming_caller_streaming_compatible_callee_inline() #0 "aarch64_pstate_sm_compatible" "aarch64_pstate_sm_body" { +define i32 @streaming_compatible_locally_streaming_caller_streaming_compatible_callee_inline() "aarch64_pstate_sm_compatible" "aarch64_pstate_sm_body" { ; CHECK-LABEL: define i32 @streaming_compatible_locally_streaming_caller_streaming_compatible_callee_inline ; CHECK-SAME: () #[[ATTR5]] { ; CHECK-NEXT: entry: @@ -474,7 +474,7 @@ entry: ; [ ] SC + B -> SC ; [x] SC + B -> N + B ; [ ] SC + B -> SC + B -define i32 @streaming_compatible_locally_streaming_caller_locally_streaming_callee_inline() #0 "aarch64_pstate_sm_compatible" "aarch64_pstate_sm_body" { +define i32 @streaming_compatible_locally_streaming_caller_locally_streaming_callee_inline() "aarch64_pstate_sm_compatible" "aarch64_pstate_sm_body" { ; CHECK-LABEL: define i32 @streaming_compatible_locally_streaming_caller_locally_streaming_callee_inline ; CHECK-SAME: () #[[ATTR5]] { ; CHECK-NEXT: entry: @@ -491,7 +491,7 @@ entry: ; [ ] SC + B -> SC ; [ ] SC + B -> N + B ; [x] SC + B -> SC + B -define i32 @streaming_compatible_locally_streaming_caller_and_callee_inline() #0 "aarch64_pstate_sm_compatible" "aarch64_pstate_sm_body" { +define i32 @streaming_compatible_locally_streaming_caller_and_callee_inline() "aarch64_pstate_sm_compatible" "aarch64_pstate_sm_body" { ; CHECK-LABEL: define i32 @streaming_compatible_locally_streaming_caller_and_callee_inline ; CHECK-SAME: () #[[ATTR5]] { ; CHECK-NEXT: entry: @@ -503,7 +503,7 @@ entry: ret i32 %res } -define void @normal_callee_with_inlineasm() #0 { +define void @normal_callee_with_inlineasm() { ; CHECK-LABEL: define void @normal_callee_with_inlineasm ; CHECK-SAME: () #[[ATTR1]] { ; CHECK-NEXT: entry: @@ -515,7 +515,7 @@ entry: ret void } -define void @streaming_caller_normal_callee_with_inlineasm_dont_inline() #0 "aarch64_pstate_sm_enabled" { +define void @streaming_caller_normal_callee_with_inlineasm_dont_inline() "aarch64_pstate_sm_enabled" { ; CHECK-LABEL: define void @streaming_caller_normal_callee_with_inlineasm_dont_inline ; CHECK-SAME: () #[[ATTR2]] { ; CHECK-NEXT: entry: @@ -527,7 +527,7 @@ entry: ret void } -define i64 @normal_callee_with_intrinsic_call() #0 { +define i64 @normal_callee_with_intrinsic_call() { ; CHECK-LABEL: define i64 @normal_callee_with_intrinsic_call ; CHECK-SAME: () #[[ATTR1]] { ; CHECK-NEXT: entry: @@ -539,7 +539,7 @@ entry: ret i64 %res } -define i64 @streaming_caller_normal_callee_with_intrinsic_call_dont_inline() #0 "aarch64_pstate_sm_enabled" { +define i64 @streaming_caller_normal_callee_with_intrinsic_call_dont_inline() "aarch64_pstate_sm_enabled" { ; CHECK-LABEL: define i64 @streaming_caller_normal_callee_with_intrinsic_call_dont_inline ; CHECK-SAME: () #[[ATTR2]] { ; CHECK-NEXT: entry: @@ -553,7 +553,7 @@ entry: declare i64 @llvm.aarch64.sve.cntb(i32) -define i64 @normal_callee_call_sme_state() #0 { +define i64 @normal_callee_call_sme_state() { ; CHECK-LABEL: define i64 @normal_callee_call_sme_state ; CHECK-SAME: () #[[ATTR1]] { ; CHECK-NEXT: entry: @@ -569,7 +569,7 @@ entry: declare {i64, i64} @__arm_sme_state() -define i64 @streaming_caller_normal_callee_call_sme_state_dont_inline() #0 "aarch64_pstate_sm_enabled" { +define i64 @streaming_caller_normal_callee_call_sme_state_dont_inline() "aarch64_pstate_sm_enabled" { ; CHECK-LABEL: define i64 @streaming_caller_normal_callee_call_sme_state_dont_inline ; CHECK-SAME: () #[[ATTR2]] { ; CHECK-NEXT: entry: @@ -585,7 +585,7 @@ entry: declare void @nonstreaming_body() -define void @nonstreaming_caller_single_nonstreaming_callee() #0 { +define void @nonstreaming_caller_single_nonstreaming_callee() { ; CHECK-LABEL: define void @nonstreaming_caller_single_nonstreaming_callee ; CHECK-SAME: () #[[ATTR1]] { ; CHECK-NEXT: call void @nonstreaming_body() @@ -595,7 +595,7 @@ define void @nonstreaming_caller_single_nonstreaming_callee() #0 { ret void } -define void @nonstreaming_caller_multiple_nonstreaming_callees() #0 { +define void @nonstreaming_caller_multiple_nonstreaming_callees() { ; CHECK-LABEL: define void @nonstreaming_caller_multiple_nonstreaming_callees ; CHECK-SAME: () #[[ATTR1]] { ; CHECK-NEXT: call void @nonstreaming_body() @@ -608,7 +608,7 @@ define void @nonstreaming_caller_multiple_nonstreaming_callees() #0 { } ; Allow inlining, as inline it would not increase the number of streaming-mode changes. -define void @streaming_caller_to_nonstreaming_callee_with_single_nonstreaming_callee_inline() #0 "aarch64_pstate_sm_enabled" { +define void @streaming_caller_to_nonstreaming_callee_with_single_nonstreaming_callee_inline() "aarch64_pstate_sm_enabled" { ; CHECK-LABEL: define void @streaming_caller_to_nonstreaming_callee_with_single_nonstreaming_callee_inline ; CHECK-SAME: () #[[ATTR2]] { ; CHECK-NEXT: call void @nonstreaming_body() @@ -619,7 +619,7 @@ define void @streaming_caller_to_nonstreaming_callee_with_single_nonstreaming_ca } ; Prevent inlining, as inlining it would lead to multiple streaming-mode changes. -define void @streaming_caller_to_nonstreaming_callee_with_multiple_nonstreaming_callees_dont_inline() #0 "aarch64_pstate_sm_enabled" { +define void @streaming_caller_to_nonstreaming_callee_with_multiple_nonstreaming_callees_dont_inline() "aarch64_pstate_sm_enabled" { ; CHECK-LABEL: define void @streaming_caller_to_nonstreaming_callee_with_multiple_nonstreaming_callees_dont_inline ; CHECK-SAME: () #[[ATTR2]] { ; CHECK-NEXT: call void @streaming_caller_to_nonstreaming_callee_with_multiple_nonstreaming_callees_dont_inline() @@ -631,7 +631,7 @@ define void @streaming_caller_to_nonstreaming_callee_with_multiple_nonstreaming_ declare void @streaming_compatible_body() "aarch64_pstate_sm_compatible" -define void @nonstreaming_caller_single_streaming_compatible_callee() #0 { +define void @nonstreaming_caller_single_streaming_compatible_callee() { ; CHECK-LABEL: define void @nonstreaming_caller_single_streaming_compatible_callee ; CHECK-SAME: () #[[ATTR1]] { ; CHECK-NEXT: call void @streaming_compatible_body() @@ -641,42 +641,75 @@ define void @nonstreaming_caller_single_streaming_compatible_callee() #0 { ret void } -define void @nonstreaming_caller_multiple_streaming_compatible_callees() #0 { -; CHECK-LABEL: define void @nonstreaming_caller_multiple_streaming_compatible_callees -; CHECK-SAME: () #[[ATTR1]] { -; CHECK-NEXT: call void @streaming_compatible_body() +; Reject inlining; while inlining would remove a streaming-mode change, it may also change runtime behaviour. +define void @streaming_caller_to_nonstreaming_callee_with_single_streamingcompatible_callee_dont_inline() "aarch64_pstate_sm_enabled" { +; CHECK-LABEL: define void @streaming_caller_to_nonstreaming_callee_with_single_streamingcompatible_callee_dont_inline +; CHECK-SAME: () #[[ATTR2]] { +; CHECK-NEXT: call void @nonstreaming_caller_single_streaming_compatible_callee() +; CHECK-NEXT: ret void +; + call void @nonstreaming_caller_single_streaming_compatible_callee() + ret void +} + +define void @nonstreaming_caller_single_streaming_compatible_callee_alwaysinline() alwaysinline { +; CHECK-LABEL: define void @nonstreaming_caller_single_streaming_compatible_callee_alwaysinline +; CHECK-SAME: () #[[ATTR6:[0-9]+]] { ; CHECK-NEXT: call void @streaming_compatible_body() ; CHECK-NEXT: ret void ; - call void @streaming_compatible_body() call void @streaming_compatible_body() ret void } -; Allow inlining, as inline would remove a streaming-mode change. -define void @streaming_caller_to_nonstreaming_callee_with_single_streamingcompatible_callee_inline() #0 "aarch64_pstate_sm_enabled" { -; CHECK-LABEL: define void @streaming_caller_to_nonstreaming_callee_with_single_streamingcompatible_callee_inline +; Reject inlining even when forced; it is unclear what the user's intentions were. +define void @streaming_caller_to_nonstreaming_alwaysinline_callee_with_single_streamingcompatible_callee_dont_inline() "aarch64_pstate_sm_enabled" { +; CHECK-LABEL: define void @streaming_caller_to_nonstreaming_alwaysinline_callee_with_single_streamingcompatible_callee_dont_inline ; CHECK-SAME: () #[[ATTR2]] { -; CHECK-NEXT: call void @streaming_compatible_body() +; CHECK-NEXT: call void @nonstreaming_caller_single_streaming_compatible_callee_alwaysinline() ; CHECK-NEXT: ret void ; - call void @nonstreaming_caller_single_streaming_compatible_callee() + call void @nonstreaming_caller_single_streaming_compatible_callee_alwaysinline() ret void } -; Allow inlining, as inline would remove several streaming-mode changes. -define void @streaming_caller_to_nonstreaming_callee_with_multiple_streamingcompatible_callees_inline() #0 "aarch64_pstate_sm_enabled" { -; CHECK-LABEL: define void @streaming_caller_to_nonstreaming_callee_with_multiple_streamingcompatible_callees_inline -; CHECK-SAME: () #[[ATTR2]] { -; CHECK-NEXT: call void @streaming_compatible_body() -; CHECK-NEXT: call void @streaming_compatible_body() +define void @invoke_opaque_fptr(ptr %fptr) alwaysinline personality ptr null { +; CHECK-LABEL: define void @invoke_opaque_fptr +; CHECK-SAME: (ptr [[FPTR:%.*]]) #[[ATTR6]] personality ptr null { +; CHECK-NEXT: invoke void [[FPTR]]() #[[ATTR13:[0-9]+]] +; CHECK-NEXT: to label [[NORMAL_RETURN:%.*]] unwind label [[UNWIND_CLEANUP:%.*]] +; CHECK: normal_return: +; CHECK-NEXT: ret void +; CHECK: unwind_cleanup: +; CHECK-NEXT: [[EH_INFO:%.*]] = landingpad { ptr, i32 } +; CHECK-NEXT: cleanup +; CHECK-NEXT: resume { ptr, i32 } [[EH_INFO]] +; + invoke void %fptr() "aarch64_pstate_sm_compatible" + to label %normal_return unwind label %unwind_cleanup + +normal_return: + ret void + +unwind_cleanup: + %eh_info = landingpad { ptr, i32 } + cleanup + resume { ptr, i32 } %eh_info +} + +; Test that we don't inline '@invoke_opaque_fptr_callee' as it contains a streaming compatible callee, +; but this time the callee is opaque pointer and called via 'invoke' rather than 'call'. +define void @callee_has_streaming_compatible_invoke_call_dont_inline(ptr %fptr) "aarch64_pstate_sm_enabled" { +; CHECK-LABEL: define void @callee_has_streaming_compatible_invoke_call_dont_inline +; CHECK-SAME: (ptr [[FPTR:%.*]]) #[[ATTR2]] { +; CHECK-NEXT: call void @invoke_opaque_fptr(ptr [[FPTR]]) ; CHECK-NEXT: ret void ; - call void @nonstreaming_caller_multiple_streaming_compatible_callees() + call void @invoke_opaque_fptr(ptr %fptr) ret void } -define void @simple_streaming_function(ptr %ptr) #0 "aarch64_pstate_sm_enabled" { +define void @simple_streaming_function(ptr %ptr) "aarch64_pstate_sm_enabled" { ; CHECK-LABEL: define void @simple_streaming_function ; CHECK-SAME: (ptr [[PTR:%.*]]) #[[ATTR2]] { ; CHECK-NEXT: store <vscale x 4 x i32> zeroinitializer, ptr [[PTR]], align 16 @@ -687,7 +720,7 @@ define void @simple_streaming_function(ptr %ptr) #0 "aarch64_pstate_sm_enabled" } ; Don't allow inlining a streaming function into a non-streaming function. -define void @non_streaming_caller_streaming_callee_dont_inline(ptr %ptr) #0 { +define void @non_streaming_caller_streaming_callee_dont_inline(ptr %ptr) { ; CHECK-LABEL: define void @non_streaming_caller_streaming_callee_dont_inline ; CHECK-SAME: (ptr [[PTR:%.*]]) #[[ATTR1]] { ; CHECK-NEXT: call void @simple_streaming_function(ptr [[PTR]]) @@ -697,7 +730,7 @@ define void @non_streaming_caller_streaming_callee_dont_inline(ptr %ptr) #0 { ret void } -define void @simple_locally_streaming_function(ptr %ptr) #0 "aarch64_pstate_sm_body" { +define void @simple_locally_streaming_function(ptr %ptr) "aarch64_pstate_sm_body" { ; CHECK-LABEL: define void @simple_locally_streaming_function ; CHECK-SAME: (ptr [[PTR:%.*]]) #[[ATTR3]] { ; CHECK-NEXT: store <vscale x 4 x i32> zeroinitializer, ptr [[PTR]], align 16 @@ -708,7 +741,7 @@ define void @simple_locally_streaming_function(ptr %ptr) #0 "aarch64_pstate_sm_b } ; Don't allow inlining a locally-streaming function into a non-streaming function. -define void @non_streaming_caller_locally_streaming_callee_dont_inline(ptr %ptr) #0 { +define void @non_streaming_caller_locally_streaming_callee_dont_inline(ptr %ptr) { ; CHECK-LABEL: define void @non_streaming_caller_locally_streaming_callee_dont_inline ; CHECK-SAME: (ptr [[PTR:%.*]]) #[[ATTR1]] { ; CHECK-NEXT: call void @simple_locally_streaming_function(ptr [[PTR]]) @@ -718,4 +751,355 @@ define void @non_streaming_caller_locally_streaming_callee_dont_inline(ptr %ptr) ret void } -attributes #0 = { "target-features"="+sve,+sme" } +; +; Don't inline when there are vscale-dependent operations +; + +; It is not safe to inline functions that have vscale-dependent operations in their body +; when the streaming modes don't match up, unless the interface of the callee takes a +; vl-dependent argument. +define ptr @vscale_dependent_op(ptr %p, i64 %k) { +; CHECK-LABEL: define ptr @vscale_dependent_op +; CHECK-SAME: (ptr [[P:%.*]], i64 [[K:%.*]]) #[[ATTR1]] { +; CHECK-NEXT: [[RES:%.*]] = getelementptr <vscale x 4 x i32>, ptr [[P]], i64 [[K]] +; CHECK-NEXT: ret ptr [[RES]] +; + %res = getelementptr <vscale x 4 x i32>, ptr %p, i64 %k + ret ptr %res +} + +define ptr @vscale_dependent_op_vl_dependent_args(ptr %p, i64 %k, <vscale x 4 x i32> %other) { +; CHECK-LABEL: define ptr @vscale_dependent_op_vl_dependent_args +; CHECK-SAME: (ptr [[P:%.*]], i64 [[K:%.*]], <vscale x 4 x i32> [[OTHER:%.*]]) #[[ATTR1]] { +; CHECK-NEXT: [[RES:%.*]] = getelementptr <vscale x 4 x i32>, ptr [[P]], i64 [[K]] +; CHECK-NEXT: store <vscale x 4 x i32> [[OTHER]], ptr [[RES]], align 16 +; CHECK-NEXT: ret ptr [[RES]] +; + %res = getelementptr <vscale x 4 x i32>, ptr %p, i64 %k + store <vscale x 4 x i32> %other, ptr %res + ret ptr %res +} + +define void @vscale_dependent_operations(ptr %p, ptr %res1ptr, ptr %res2ptr) "aarch64_pstate_sm_enabled" { +; CHECK-LABEL: define void @vscale_dependent_operations +; CHECK-SAME: (ptr [[P:%.*]], ptr [[RES1PTR:%.*]], ptr [[RES2PTR:%.*]]) #[[ATTR2]] { +; CHECK-NEXT: [[RES1:%.*]] = call ptr @vscale_dependent_op(ptr [[P]], i64 4) +; CHECK-NEXT: [[RES_I:%.*]] = getelementptr <vscale x 4 x i32>, ptr [[P]], i64 4 +; CHECK-NEXT: store <vscale x 4 x i32> zeroinitializer, ptr [[RES_I]], align 16 +; CHECK-NEXT: store ptr [[RES1]], ptr [[RES1PTR]], align 8 +; CHECK-NEXT: store ptr [[RES_I]], ptr [[RES2PTR]], align 8 +; CHECK-NEXT: ret void +; + %res1 = call ptr @vscale_dependent_op(ptr %p, i64 4) + %res2 = call ptr @vscale_dependent_op_vl_dependent_args(ptr %p, i64 4, <vscale x 4 x i32> zeroinitializer) + store ptr %res1, ptr %res1ptr + store ptr %res2, ptr %res2ptr + ret void +} + +; functions with scalable alloca's shouldn't be inlined if the streaming properties don't match +; as a scalable alloca is a vl-dependent operation. +define void @scalable_alloca_op(ptr %p) { +; CHECK-LABEL: define void @scalable_alloca_op +; CHECK-SAME: (ptr [[P:%.*]]) #[[ATTR1]] { +; CHECK-NEXT: [[ALLOCA:%.*]] = alloca <vscale x 4 x i32>, align 1 +; CHECK-NEXT: call void [[P]](ptr [[ALLOCA]]) +; CHECK-NEXT: ret void +; + %alloca = alloca <vscale x 4 x i32>, align 1 + call void %p(ptr %alloca) + ret void +} + +define void @incompatible_scalable_alloca_op_caller(ptr %p) "aarch64_pstate_sm_enabled" { +; CHECK-LABEL: define void @incompatible_scalable_alloca_op_caller +; CHECK-SAME: (ptr [[P:%.*]]) #[[ATTR2]] { +; CHECK-NEXT: call void @scalable_alloca_op(ptr [[P]]) +; CHECK-NEXT: ret void +; + call void @scalable_alloca_op(ptr %p) + ret void +} + +; Generic intrinsics that take/return a scalable type, are vscale-dependent operations. +define i64 @intrinsic_with_scalable_type(ptr %p) { +; CHECK-LABEL: define i64 @intrinsic_with_scalable_type +; CHECK-SAME: (ptr [[P:%.*]]) #[[ATTR1]] { +; CHECK-NEXT: [[LD:%.*]] = load <vscale x 2 x i64>, ptr [[P]], align 16 +; CHECK-NEXT: [[RES:%.*]] = call i64 @llvm.vector.reduce.add.nxv2i64(<vscale x 2 x i64> [[LD]]) +; CHECK-NEXT: ret i64 [[RES]] +; + %ld = load <vscale x 2 x i64>, ptr %p + %res = call i64 @llvm.vector.reduce.add(<vscale x 2 x i64> %ld) + ret i64 %res +} + +define i64 @intrinsic_with_scalable_type_vscale_must_match(ptr %p, <vscale x 4 x i32> %unused) { +; CHECK-LABEL: define i64 @intrinsic_with_scalable_type_vscale_must_match +; CHECK-SAME: (ptr [[P:%.*]], <vscale x 4 x i32> [[UNUSED:%.*]]) #[[ATTR1]] { +; CHECK-NEXT: [[LD:%.*]] = load <vscale x 2 x i64>, ptr [[P]], align 16 +; CHECK-NEXT: [[RES:%.*]] = call i64 @llvm.vector.reduce.add.nxv2i64(<vscale x 2 x i64> [[LD]]) +; CHECK-NEXT: ret i64 [[RES]] +; + %ld = load <vscale x 2 x i64>, ptr %p + %res = call i64 @llvm.vector.reduce.add(<vscale x 2 x i64> %ld) + ret i64 %res +} + +define i64 @generic_intrinsic_with_scalable_types(ptr %p) "aarch64_pstate_sm_enabled" { +; CHECK-LABEL: define i64 @generic_intrinsic_with_scalable_types +; CHECK-SAME: (ptr [[P:%.*]]) #[[ATTR2]] { +; CHECK-NEXT: [[RES1:%.*]] = call i64 @intrinsic_with_scalable_type(ptr [[P]]) +; CHECK-NEXT: [[LD_I:%.*]] = load <vscale x 2 x i64>, ptr [[P]], align 16 +; CHECK-NEXT: [[RES_I:%.*]] = call i64 @llvm.vector.reduce.add.nxv2i64(<vscale x 2 x i64> [[LD_I]]) +; CHECK-NEXT: [[RES:%.*]] = add i64 [[RES1]], [[RES_I]] +; CHECK-NEXT: ret i64 [[RES]] +; + %res1 = call i64 @intrinsic_with_scalable_type(ptr %p) + %res2 = call i64 @intrinsic_with_scalable_type_vscale_must_match(ptr %p, <vscale x 4 x i32> zeroinitializer) + %res = add i64 %res1, %res2 + ret i64 %res +} + +; +; Don't inline inline-asm when streaming-modes are incompatible, as the asm may +; contain instructions that are invalid or may behave diff erently in the mode of +; the caller. +; + +define void @inline_asm() { +; CHECK-LABEL: define void @inline_asm +; CHECK-SAME: () #[[ATTR1]] { +; CHECK-NEXT: call void asm sideeffect "", ""() +; CHECK-NEXT: ret void +; + call void asm sideeffect "", ""() + ret void +} + +define void @incompatible_inline_asm_sm() "aarch64_pstate_sm_enabled" { +; CHECK-LABEL: define void @incompatible_inline_asm_sm +; CHECK-SAME: () #[[ATTR2]] { +; CHECK-NEXT: call void @inline_asm() +; CHECK-NEXT: ret void +; + call void @inline_asm() + ret void +} + +; +; Don't inline functions that contain calls to SME ABI routines (like __arm_get_current_vg()) +; + +declare i64 @__arm_get_current_vg() + +define i64 @current_vg() { +; CHECK-LABEL: define i64 @current_vg +; CHECK-SAME: () #[[ATTR1]] { +; CHECK-NEXT: [[VSCALE:%.*]] = call i64 @__arm_get_current_vg() +; CHECK-NEXT: ret i64 [[VSCALE]] +; + %vscale = call i64 @__arm_get_current_vg() + ret i64 %vscale +} + +define i64 @compatible_current_vg_caller() { +; CHECK-LABEL: define i64 @compatible_current_vg_caller +; CHECK-SAME: () #[[ATTR1]] { +; CHECK-NEXT: [[VSCALE_I:%.*]] = call i64 @__arm_get_current_vg() +; CHECK-NEXT: ret i64 [[VSCALE_I]] +; + %vscale = call i64 @current_vg() + ret i64 %vscale +} + +define i64 @incompatible_current_vg_caller() "aarch64_pstate_sm_enabled" { +; CHECK-LABEL: define i64 @incompatible_current_vg_caller +; CHECK-SAME: () #[[ATTR2]] { +; CHECK-NEXT: [[VSCALE:%.*]] = call i64 @current_vg() +; CHECK-NEXT: ret i64 [[VSCALE]] +; + %vscale = call i64 @current_vg() + ret i64 %vscale +} + +; +; Don't inline NEON intrinsics into a streaming mode functions. +; + +define i32 @neon_intrinsic(<4 x i32> %in) { +; CHECK-LABEL: define i32 @neon_intrinsic +; CHECK-SAME: (<4 x i32> [[IN:%.*]]) #[[ATTR1]] { +; CHECK-NEXT: [[RES:%.*]] = call i32 @llvm.aarch64.neon.uaddv.i32.v4i32(<4 x i32> [[IN]]) +; CHECK-NEXT: ret i32 [[RES]] +; + %res = call i32 @llvm.aarch64.neon.uaddv.i32.v4i32(<4 x i32> %in) + ret i32 %res +} + +define i32 @incompatible_neon_intrinsic_caller(<4 x i32> %in) "aarch64_pstate_sm_enabled" { +; CHECK-LABEL: define i32 @incompatible_neon_intrinsic_caller +; CHECK-SAME: (<4 x i32> [[IN:%.*]]) #[[ATTR2]] { +; CHECK-NEXT: [[VSCALE:%.*]] = call i32 @neon_intrinsic(<4 x i32> [[IN]]) +; CHECK-NEXT: ret i32 [[VSCALE]] +; + %vscale = call i32 @neon_intrinsic(<4 x i32> %in) + ret i32 %vscale +} + +; a bit of a niche case, but if the caller uses ZA but is not in streaming-mode, a NEON intrinsic is safe. +define i32 @compatible_neon_intrinsic_caller_za(<4 x i32> %in) "aarch64_pstate_za_enabled" { +; CHECK-LABEL: define i32 @compatible_neon_intrinsic_caller_za +; CHECK-SAME: (<4 x i32> [[IN:%.*]]) #[[ATTR7:[0-9]+]] { +; CHECK-NEXT: [[RES_I:%.*]] = call i32 @llvm.aarch64.neon.uaddv.i32.v4i32(<4 x i32> [[IN]]) +; CHECK-NEXT: ret i32 [[RES_I]] +; + %vscale = call i32 @neon_intrinsic(<4 x i32> %in) + ret i32 %vscale +} + +; +; Don't inline fixed-length vector code, as performance may be affected. +; + +; Functions with fixed-length vectors shouldn't be inlined if the streaming properties don't match +; as performance may be affected. However, when they have the alwaysinline property, they should +; still be inlined. + +define void @fixed_length_vector_operation_alwaysinline(ptr %p) alwaysinline { +; CHECK-LABEL: define void @fixed_length_vector_operation_alwaysinline +; CHECK-SAME: (ptr [[P:%.*]]) #[[ATTR6]] { +; CHECK-NEXT: store <4 x i32> zeroinitializer, ptr [[P]], align 16 +; CHECK-NEXT: ret void +; + store <4 x i32> zeroinitializer, ptr %p + ret void +} + +define void @fixed_length_vector_operation(ptr %p) { +; CHECK-LABEL: define void @fixed_length_vector_operation +; CHECK-SAME: (ptr [[P:%.*]]) #[[ATTR1]] { +; CHECK-NEXT: store <4 x i32> zeroinitializer, ptr [[P]], align 16 +; CHECK-NEXT: ret void +; + store <4 x i32> zeroinitializer, ptr %p + ret void +} + +define void @fixed_length_vector_operations(ptr %p) "aarch64_pstate_sm_enabled" { +; CHECK-LABEL: define void @fixed_length_vector_operations +; CHECK-SAME: (ptr [[P:%.*]]) #[[ATTR2]] { +; CHECK-NEXT: store <4 x i32> zeroinitializer, ptr [[P]], align 16 +; CHECK-NEXT: call void @fixed_length_vector_operation(ptr [[P]]) +; CHECK-NEXT: ret void +; + call void @fixed_length_vector_operation_alwaysinline(ptr %p) + call void @fixed_length_vector_operation(ptr %p) + ret void +} + +; +; strictfp and streaming-mode +; + +define float @strict_fp(float %in) strictfp { +; CHECK-LABEL: define float @strict_fp +; CHECK-SAME: (float [[IN:%.*]]) #[[ATTR8:[0-9]+]] { +; CHECK-NEXT: [[RES:%.*]] = fadd float [[IN]], 4.200000e+01 +; CHECK-NEXT: ret float [[RES]] +; + %res = fadd float %in, 42.0; + ret float %res +} + +; floating point environment is diff erent in streaming mode, so don't inline. +define float @incompatible_fp_environment_sm(float %in) strictfp "aarch64_pstate_sm_enabled" { +; CHECK-LABEL: define float @incompatible_fp_environment_sm +; CHECK-SAME: (float [[IN:%.*]]) #[[ATTR9:[0-9]+]] { +; CHECK-NEXT: [[RES:%.*]] = call float @strict_fp(float [[IN]]) +; CHECK-NEXT: ret float [[RES]] +; + %res = call float @strict_fp(float %in) + ret float %res +} + +; +; Test we can inline fixed-length gather/scatter operations into streaming functions, +; as those can be scalarized by the code-generator. +; + +define void @scatter_fixed_length(ptr %ptr_to_ptrs, ptr %ptr_to_vals) alwaysinline { +; CHECK-LABEL: define void @scatter_fixed_length +; CHECK-SAME: (ptr [[PTR_TO_PTRS:%.*]], ptr [[PTR_TO_VALS:%.*]]) #[[ATTR6]] { +; CHECK-NEXT: [[PTRS:%.*]] = load <2 x ptr>, ptr [[PTR_TO_PTRS]], align 16 +; CHECK-NEXT: [[VALS:%.*]] = load <2 x i64>, ptr [[PTR_TO_VALS]], align 16 +; CHECK-NEXT: call void @llvm.masked.scatter.v2i64.v2p0(<2 x i64> [[VALS]], <2 x ptr> align 8 [[PTRS]], <2 x i1> splat (i1 true)) +; CHECK-NEXT: ret void +; + %ptrs = load <2 x ptr>, ptr %ptr_to_ptrs + %vals = load <2 x i64>, ptr %ptr_to_vals + call void @llvm.masked.scatter(<2 x i64> %vals, <2 x ptr> %ptrs, i32 8, <2 x i1> splat(i1 true)) + ret void +} + +define void @scatter_fixed_length_caller(ptr %ptr_to_ptrs, ptr %ptr_to_vals) "aarch64_pstate_sm_enabled" { +; CHECK-LABEL: define void @scatter_fixed_length_caller +; CHECK-SAME: (ptr [[PTR_TO_PTRS:%.*]], ptr [[PTR_TO_VALS:%.*]]) #[[ATTR2]] { +; CHECK-NEXT: [[PTRS_I:%.*]] = load <2 x ptr>, ptr [[PTR_TO_PTRS]], align 16 +; CHECK-NEXT: [[VALS_I:%.*]] = load <2 x i64>, ptr [[PTR_TO_VALS]], align 16 +; CHECK-NEXT: call void @llvm.masked.scatter.v2i64.v2p0(<2 x i64> [[VALS_I]], <2 x ptr> align 8 [[PTRS_I]], <2 x i1> splat (i1 true)) +; CHECK-NEXT: ret void +; + call void @scatter_fixed_length(ptr %ptr_to_ptrs, ptr %ptr_to_vals) + ret void +} + +; +; Test that we never inline a scalable gather/scatter operation into a streaming function. +; + +define void @scatter_scalable_length(<vscale x 2 x ptr> %ptrs, <vscale x 2 x i64> %vals) alwaysinline { +; CHECK-LABEL: define void @scatter_scalable_length +; CHECK-SAME: (<vscale x 2 x ptr> [[PTRS:%.*]], <vscale x 2 x i64> [[VALS:%.*]]) #[[ATTR6]] { +; CHECK-NEXT: call void @llvm.masked.scatter.nxv2i64.nxv2p0(<vscale x 2 x i64> [[VALS]], <vscale x 2 x ptr> align 8 [[PTRS]], <vscale x 2 x i1> splat (i1 true)) +; CHECK-NEXT: ret void +; + call void @llvm.masked.scatter(<vscale x 2 x i64> %vals, <vscale x 2 x ptr> %ptrs, i32 8, <vscale x 2 x i1> splat(i1 true)) + ret void +} + +define void @scatter_scalable_length_caller(<vscale x 2 x ptr> %ptrs, <vscale x 2 x i64> %vals) "aarch64_pstate_sm_enabled" { +; CHECK-LABEL: define void @scatter_scalable_length_caller +; CHECK-SAME: (<vscale x 2 x ptr> [[PTRS:%.*]], <vscale x 2 x i64> [[VALS:%.*]]) #[[ATTR2]] { +; CHECK-NEXT: call void @scatter_scalable_length(<vscale x 2 x ptr> [[PTRS]], <vscale x 2 x i64> [[VALS]]) +; CHECK-NEXT: ret void +; + call void @scatter_scalable_length(<vscale x 2 x ptr> %ptrs, <vscale x 2 x i64> %vals) + ret void +} + +; +; Test that we inline a function with a call to llvm.vscale() (which takes/returns +; no scalable vectors) when the function has vl-dependent arguments. The case for +; non-vl-dependent arguments is already tested above. +; + +define i64 @llvm_vscale_vl_dependent_args(<vscale x 2 x i64> %in) { +; CHECK-LABEL: define i64 @llvm_vscale_vl_dependent_args +; CHECK-SAME: (<vscale x 2 x i64> [[IN:%.*]]) #[[ATTR1]] { +; CHECK-NEXT: [[RES:%.*]] = call i64 @llvm.vscale.i64() +; CHECK-NEXT: ret i64 [[RES]] +; + %res = call i64 @llvm.vscale.i64() + ret i64 %res +} + +define i64 @llvm_vscale_vl_dependent_args_caller() "aarch64_pstate_sm_enabled" { +; CHECK-LABEL: define i64 @llvm_vscale_vl_dependent_args_caller +; CHECK-SAME: () #[[ATTR2]] { +; CHECK-NEXT: [[RES_I:%.*]] = call i64 @llvm.vscale.i64() +; CHECK-NEXT: ret i64 [[RES_I]] +; + %res = call i64 @llvm_vscale_vl_dependent_args(<vscale x 2 x i64> zeroinitializer) + ret i64 %res +} diff --git a/llvm/test/Transforms/Inline/AArch64/sme-pstatesm-inline-transitive.ll b/llvm/test/Transforms/Inline/AArch64/sme-pstatesm-inline-transitive.ll new file mode 100644 index 00000000000000..808749251ce0ab --- /dev/null +++ b/llvm/test/Transforms/Inline/AArch64/sme-pstatesm-inline-transitive.ll @@ -0,0 +1,47 @@ +; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 2 +; RUN: opt -mattr=+sve,+sme -S -passes=inline -inlinedefault-threshold=1 < %s | FileCheck %s --check-prefixes=CHECK,TH1 +; RUN: opt -mattr=+sve,+sme -S -passes=inline -inlinedefault-threshold=25 < %s | FileCheck %s --check-prefixes=CHECK,TH25 + +target triple = "aarch64" + +declare void @streaming_f() "aarch64_pstate_sm_enabled" + +define void @c() "aarch64_pstate_sm_enabled" { +; CHECK-LABEL: define void @c +; CHECK-SAME: () #[[ATTR1:[0-9]+]] { +; CHECK-NEXT: call void @streaming_f() +; CHECK-NEXT: call void @streaming_f() +; CHECK-NEXT: ret void +; + call void @streaming_f() + call void @streaming_f() + ret void +} + +; Don't inline the call to @c, as inlining a streaming function into a non-streaming function is considered incompatible. +define void @b() { +; CHECK-LABEL: define void @b +; CHECK-SAME: () #[[ATTR2:[0-9]+]] { +; CHECK-NEXT: call void @c() +; CHECK-NEXT: ret void +; + call void @c() + ret void +} + +; Inline the call to @c in @a, as that avoids streaming mode changes. +define void @a() "aarch64_pstate_sm_enabled" { +; TH1-LABEL: define void @a +; TH1-SAME: () #[[ATTR1]] { +; TH1-NEXT: call void @c() +; TH1-NEXT: ret void +; +; TH25-LABEL: define void @a +; TH25-SAME: () #[[ATTR1]] { +; TH25-NEXT: call void @streaming_f() +; TH25-NEXT: call void @streaming_f() +; TH25-NEXT: ret void +; + call void @b() + ret void +} _______________________________________________ cfe-commits mailing list [email protected] https://lists.llvm.org/cgi-bin/mailman/listinfo/cfe-commits
