diff --git a/examples/llvm/IR/Intrinsics.td b/examples/llvm/IR/Intrinsics.td index 993ddd7..a376d4a 100644 --- a/examples/llvm/IR/Intrinsics.td +++ b/examples/llvm/IR/Intrinsics.td @@ -59,6 +59,7 @@ class IntrinsicMemoryLocation; // TODO: Populate with all IRMemLocation enum values and update // getValueAsIRMemLocation accordingly. +def ArgMem : IntrinsicMemoryLocation; def InaccessibleMem : IntrinsicMemoryLocation; def TargetMem0 : IntrinsicMemoryLocation; def TargetMem1 : IntrinsicMemoryLocation; @@ -73,6 +74,11 @@ class IntrWrite idx> : IntrinsicProperty { list MemLoc=idx; } +// Constrain intrinsic to not write any memory location. +defvar IntrReadOnly = IntrWrite<[]>; +// Constrain intrinsic to not read any memory location. +defvar IntrWriteOnly = IntrRead<[]>; + // Commutative - This intrinsic is commutative: X op Y == Y op X. def Commutative : IntrinsicProperty; @@ -130,9 +136,21 @@ class Returned : IntrinsicProperty { int ArgNo = idx.Value; } -// ImmArg - The specified argument must be an immediate. -class ImmArg : IntrinsicProperty { +// Default value for a trailing ImmArg, materialized by AutoUpgrade. +class DefaultValue { + int Value = val; +} + +// Sentinel — used as the default for ImmArg's optional DefaultValue parameter. +def NoDefault : DefaultValue<0> { + let Value = ?; +} + +// ImmArg - The specified argument must be an immediate. The optional +// second template parameter specifies a default value for AutoUpgrade. +class ImmArg : IntrinsicProperty { int ArgNo = idx.Value; + DefaultValue Default = val; } // ReadOnly - The specified argument pointer is not written to through the @@ -153,6 +171,25 @@ class ReadNone : IntrinsicProperty { int ArgNo = idx.Value; } +// A RangeSet specifies a union of ordered and disjoint closed ranges that an +// immediate argument must be in. +class RangeSet> ranges> : IntrinsicProperty { + list> ValidRanges = !filter(r, ranges, !eq(!size(r), 2)); + + assert !not(!empty(ranges)), "RangeSet requires at least one range"; + assert !eq(!size(ValidRanges), !size(ranges)), + "RangeSet requires each range to have exactly two bounds"; + assert !empty(!filter(r, ValidRanges, !lt(r[1], r[0]))), + "RangeSet requires lower <= upper for each range"; + assert !empty(!filter(i, !range(1, !size(ValidRanges)), + !le(ValidRanges[i][0], + ValidRanges[!sub(i, 1)][1]))), + "RangeSet requires ordered and non-overlapping ranges"; + + int ArgNo = idx.Value; + list> Ranges = ranges; +} + // The return value or argument is in the range [lower, upper), // where lower and upper are interpreted as signed integers. class Range : IntrinsicProperty { @@ -161,7 +198,13 @@ class Range : IntrinsicProperty { int Upper = upper; } -// ArgProperty - Base class for argument properties that can be specified in ArgInfo. +// The underlying object of the argument/return can not be freed. +class NoFreeObj : IntrinsicProperty { + int ArgNo = idx.Value; +} + +// ArgProperty - Base class for argument properties that can be specified in +// ArgInfo. class ArgProperty; // ArgName - Specifies the name of an argument for pretty-printing. @@ -174,10 +217,11 @@ class ImmArgPrinter : ArgProperty { string FuncName = funcname; } -// ArgInfo - The specified argument has properties defined by a list of ArgProperty objects. -class ArgInfo arg_properties> : IntrinsicProperty { +// ArgInfo - The specified argument has properties defined by a list of +// ArgProperty objects. +class ArgInfo properties> : IntrinsicProperty { int ArgNo = idx.Value; - list Properties = arg_properties; + list Properties = properties; } def IntrNoReturn : IntrinsicProperty; @@ -235,39 +279,36 @@ def IntrTriviallyScalarizable : IntrinsicProperty; // IIT constants and utils //===----------------------------------------------------------------------===// -// llvm::Intrinsic::IITDescriptor::AnyKind::AK_% -def AnyKind { - int Any = 0; - int AnyInteger = 1; - int AnyFloat = 2; - int AnyVector = 3; - int AnyPointer = 4; - - int MatchType = 7; +// llvm::Intrinsic::IITDescriptor::AnyKindVectorConstraint::VC_% +def AnyKindVectorConstraint { + int None = 0; + int Vector = 1; + int Scalar = 2; } -// Placeholder to encode the overload index of the current type. We encode bit -// 8 = 1 to indicate that this entry needs to be patched up with the overload -// index (to prevent conflict with any valid not-to-be-patched IIT enccoding -// byte, whose value will be <= 255). The AnyKind itself is in the lower bits. -// Note that this is just a transient representation till its gets processed -// by `DoPatchOverloadIndex` below, so this is *not* the encoding of the -// final type signature. -class OverloadIndexPlaceholder { - int ID = 0x100; - int ret = !or(ID, AnyKindVal); +// llvm::Intrinsic::IITDescriptor::AnyKindElementConstraint::EC_% +def AnyKindElementConstraint { + int None = 0; + int Integer = 1; + int Float = 2; + int Pointer = 3; } -// This class defines how the overload index and the arg kind are actually -// packed into a single byte for the final IIT table encoding. Overload index is -// packed in low 5 bits, argument kind is packed in upper 3 bits. This enables -// us to use the same packing for llvm_any* types, which use a argument kind -// and for partially dependent types like `LLVMVectorOfAnyPointersToElt` which -// do not use the argument kind and expect the overload index in the lower bits. -class PackOverloadIndex { - assert !lt(OverloadIndex, 32), "Cannot support more than 32 overload types"; - assert !lt(AnyKindVal, 8), "Cannot support more than 8 argument kinds"; - int ret = !or(!shl(AnyKindVal, 5), OverloadIndex); +// Placeholder to encode the overload index of the current type. We encode a +// value > 255 to indicate that this entry needs to be patched up with the +// overload index (to prevent conflict with any valid not-to-be-patched IIT +// encoding byte, whose value will be <= 255). Note that this is just a +// transient representation till it gets processed by `PatchOverloadIndex` +// below, so this is *not* the encoding of the final type signature. +defvar OverloadIndexPlaceholder = 0x100; + +// This class verifies that the overload index is valid for the final IIT table +// encoding. Overload index has a single byte assigned in the IIT encoding, so +// verify that its >= 0 and <= 255. +class VerifyOverloadIndex { + assert !ge(OverloadIndex, 0), "overload index must be >= 0"; + assert !lt(OverloadIndex, 256), "cannot support more than 256 overload types"; + int ret = OverloadIndex; } // This class handles the actual patching of the overload index into a component @@ -275,11 +316,11 @@ class PackOverloadIndex { // value generated by OverloadIndexPlaceholder and patching is needed, else the // value is left unchanged. class PatchOverloadIndex { - int AnyKindVal = !and(Sig, 0x7); - int ret = !cond( - // If the value is > 255, it indicates that patching is needed. - !gt(Sig, 255) : PackOverloadIndex.ret, - true: Sig); + // If the value is equal to `OverloadIndexPlaceholder`, it indicates that + // patching is needed by replacing it with this type's overload index. + int ret = !if(!eq(Sig, OverloadIndexPlaceholder), + VerifyOverloadIndex.ret, + Sig); } //===----------------------------------------------------------------------===// @@ -327,7 +368,7 @@ def IIT_V64 : IIT_Vec<64, 16>; def IIT_MMX : IIT_VT; def IIT_TOKEN : IIT_VT; def IIT_METADATA : IIT_VT; -// Note: Unused IIT code 20. +def IIT_MATCH : IIT_Base<20>; def IIT_STRUCT : IIT_Base<21>; def IIT_EXTEND_ARG : IIT_Base<22>; def IIT_TRUNC_ARG : IIT_Base<23>; @@ -399,17 +440,18 @@ class LLVMType { !foreach(iit, IITs, iit.Number)); } -class LLVMAnyType : LLVMType { - int ArgCode = !cond( - !eq(vt, Any) : AnyKind.Any, - !eq(vt, iAny) : AnyKind.AnyInteger, - !eq(vt, fAny) : AnyKind.AnyFloat, - !eq(vt, vAny) : AnyKind.AnyVector, - !eq(vt, pAny) : AnyKind.AnyPointer, - ); +class LLVMAnyType : LLVMType { let Sig = [ IIT_ANY.Number, - OverloadIndexPlaceholder .ret, + + // The first byte of the IIT_ANY payload is the overload index of this type. + // Here we use `OverloadIndexPlaceholder` which will be updated to the + // overload index in `PatchOverloadIndex`. + OverloadIndexPlaceholder, + + // The second byte of IIT_ANY payload is the packed AnyKindVectorConstraint + // and AnyKindElementConstraint. + !or(!shl(VecKind, 4), ElemKind) ]; assert VT.isOverloaded, "LLVMAnyType.VT should have isOverloaded"; @@ -429,9 +471,6 @@ class LLVMQualPointerType ]); } -// Note: CodeGenIntrinsics.cpp seems to check this class to check pointers. -class LLVMAnyPointerType : LLVMAnyType; - // Dependent types: These are types that depend on another LLVMAnyType overload // type. There are 2 subclasses of dependent types: // 1. Fully dependent types: dependent type can be completely derived from @@ -452,7 +491,7 @@ class LLVMFullyDependentType // followed by the overload index of the overload type it depends on. let Sig = [ IIT_Info.Number, - PackOverloadIndex.ret + VerifyOverloadIndex.ret ]; } @@ -463,10 +502,10 @@ class LLVMPartiallyDependentType // overload type its depends on. let Sig = [ IIT_Info.Number, - // This types overload index, arg kind ignored. - OverloadIndexPlaceholder <0>.ret, - // Overload index of the reference overload type, arg kind ignored. - PackOverloadIndex.ret, + // This types overload index. + OverloadIndexPlaceholder, + // Overload index of the reference overload type. + VerifyOverloadIndex.ret, ]; } @@ -492,7 +531,7 @@ class LLVMPartiallyDependentType // // LLVMMatchType<0> therefore will match the return type, and // LLVMMatchType<2> will match the 3rd argument. -class LLVMMatchType : LLVMFullyDependentType; +class LLVMMatchType : LLVMFullyDependentType; // Match the type of another intrinsic parameter that is expected to be based on // an integral type (i.e. either iN or ), but change the scalar size to @@ -508,8 +547,8 @@ class LLVMScalarOrSameVectorWidth : LLVMFullyDependentType { let Sig = !listconcat([ IIT_SAME_VEC_WIDTH_ARG.Number, - // Overload index of the reference overload type, arg kind ignored. - PackOverloadIndex.ret, + // Overload index of the reference overload type. + VerifyOverloadIndex.ret, ], elty.Sig); } @@ -522,7 +561,7 @@ class LLVMOneNthElementsVectorType : LLVMFullyDependentType { let Sig = [ IIT_ONE_NTH_ELTS_VEC_ARG.Number, - PackOverloadIndex.ret, + VerifyOverloadIndex.ret, n, ]; } @@ -549,12 +588,51 @@ class LLVMVectorOfAnyPointersToElt def llvm_void_ty : LLVMType; -def llvm_any_ty : LLVMAnyType; -def llvm_anyint_ty : LLVMAnyType; -def llvm_anyfloat_ty : LLVMAnyType; -def llvm_anyvector_ty : LLVMAnyType; +// Unconstrained overloaded type. +def llvm_any_ty : LLVMAnyType; +// Integer overloaded types. +def llvm_anyint_ty : LLVMAnyType; +def llvm_any_vector_int_ty : LLVMAnyType; +def llvm_any_scalar_int_ty : LLVMAnyType; + +// FP overloaded types. +def llvm_anyfloat_ty : LLVMAnyType; +def llvm_any_vector_float_ty : LLVMAnyType; +def llvm_any_scalar_float_ty : LLVMAnyType; + +// Pointer overload types. +// Note: CodeGenIntrinsics.cpp seems to check this class to check pointers. +class LLVMAnyPointerType : LLVMAnyType; + +// Any scalar pointer type. def llvm_anyptr_ty : LLVMAnyPointerType; // ptr addrspace(N) +// Any vector of pointers. +def llvm_any_vector_ptr_ty : LLVMAnyType; + +// Other overloaded types. +def llvm_anyvector_ty : LLVMAnyType; + def llvm_i1_ty : LLVMType; def llvm_i8_ty : LLVMType; def llvm_i16_ty : LLVMType; @@ -729,6 +807,11 @@ class TypeInfoGen RetTypes, list ParamTypes> { // * ParamTypes is a list containing the parameter types expected for the // intrinsic. // * Properties can be set to describe the behavior of the intrinsic. +// * TargetFeatures is a target feature expression required by the intrinsic. +// The empty string means no target features are required. The expression +// uses feature names from the target's subtarget feature table. Comma means +// AND, | means OR, comma has higher precedence than |, and parentheses group +// expressions. // class Intrinsic ret_types, list param_types = [], @@ -738,6 +821,7 @@ class Intrinsic ret_types, bit disable_default_attributes = true> : SDPatternOperator { string LLVMName = name; string TargetPrefix = ""; // Set to a prefix for target-specific intrinsics. + string TargetFeatures = ""; // Target features required by this intrinsic. list RetTypes = ret_types; list ParamTypes = param_types; list IntrProperties = intr_properties; @@ -760,8 +844,8 @@ class DefaultAttrsIntrinsic ret_types, intr_properties, name, sd_properties, /*disable_default_attributes*/ 0> {} -/// ClangBuiltin - If this intrinsic exactly corresponds to a Clang builtin, this -/// specifies the name of the builtin. This provides automatic CBE and CFE +/// ClangBuiltin - If this intrinsic exactly corresponds to a Clang builtin, +/// this specifies the name of the builtin. This provides automatic CBE and CFE /// support. class ClangBuiltin { string ClangBuiltinName = name; @@ -771,18 +855,28 @@ class MSBuiltin { string MSBuiltinName = name; } +/// RequiresTargetFeatures - If this intrinsic requires target features, +/// this specifies the required feature expression using feature names from the +/// target's subtarget feature table. The expression grammar matches Clang +/// builtins: comma means AND, | means OR, comma has higher precedence than |, +/// and parentheses group expressions. +class RequiresTargetFeatures { + string TargetFeatures = features; +} + /// Utility class for intrinsics that /// 1. Don't touch memory or any hidden state /// 2. Can be freely speculated, and /// 3. Will not create undef or poison on defined inputs. class PureIntrinsic ret_types, - list param_types = [], - list intr_properties = [], - string name = "", - list sd_properties = []> - : DefaultAttrsIntrinsic; + list param_types = [], + list intr_properties = [], + string name = "", + list sd_properties = []> + : DefaultAttrsIntrinsic; #ifndef TEST_INTRINSICS_SUPPRESS_DEFS @@ -814,79 +908,79 @@ def int_gcwrite : Intrinsic<[], // Note these are to support the Objective-C ARC optimizer which wants to // eliminate retain and releases where possible. -def int_objc_autorelease : Intrinsic<[llvm_ptr_ty], - [llvm_ptr_ty], - [Returned>]>; -def int_objc_autoreleasePoolPop : Intrinsic<[], [llvm_ptr_ty]>; -def int_objc_autoreleasePoolPush : Intrinsic<[llvm_ptr_ty], []>; -def int_objc_autoreleaseReturnValue : Intrinsic<[llvm_ptr_ty], - [llvm_ptr_ty], - [Returned>]>; -def int_objc_copyWeak : Intrinsic<[], - [llvm_ptr_ty, - llvm_ptr_ty]>; -def int_objc_destroyWeak : Intrinsic<[], [llvm_ptr_ty]>; -def int_objc_initWeak : Intrinsic<[llvm_ptr_ty], - [llvm_ptr_ty, - llvm_ptr_ty]>; -def int_objc_loadWeak : Intrinsic<[llvm_ptr_ty], - [llvm_ptr_ty]>; -def int_objc_loadWeakRetained : Intrinsic<[llvm_ptr_ty], - [llvm_ptr_ty]>; -def int_objc_moveWeak : Intrinsic<[], - [llvm_ptr_ty, - llvm_ptr_ty]>; - -def int_objc_release : Intrinsic<[], [llvm_ptr_ty]>; -def int_objc_retain : Intrinsic<[llvm_ptr_ty], - [llvm_ptr_ty], - [Returned>]>; - -def int_objc_retainAutorelease : Intrinsic<[llvm_ptr_ty], - [llvm_ptr_ty], - [Returned>]>; -def int_objc_retainAutoreleaseReturnValue : Intrinsic<[llvm_ptr_ty], - [llvm_ptr_ty], - [Returned>]>; +def int_objc_autorelease : Intrinsic<[llvm_ptr_ty], + [llvm_ptr_ty], + [Returned>]>; +def int_objc_autoreleasePoolPop : Intrinsic<[], [llvm_ptr_ty]>; +def int_objc_autoreleasePoolPush : Intrinsic<[llvm_ptr_ty], []>; +def int_objc_autoreleaseReturnValue : Intrinsic<[llvm_ptr_ty], + [llvm_ptr_ty], + [Returned>]>; +def int_objc_copyWeak : Intrinsic<[], + [llvm_ptr_ty, + llvm_ptr_ty]>; +def int_objc_destroyWeak : Intrinsic<[], [llvm_ptr_ty]>; +def int_objc_initWeak : Intrinsic<[llvm_ptr_ty], + [llvm_ptr_ty, + llvm_ptr_ty]>; +def int_objc_loadWeak : Intrinsic<[llvm_ptr_ty], + [llvm_ptr_ty]>; +def int_objc_loadWeakRetained : Intrinsic<[llvm_ptr_ty], + [llvm_ptr_ty]>; +def int_objc_moveWeak : Intrinsic<[], + [llvm_ptr_ty, + llvm_ptr_ty]>; + +def int_objc_release : Intrinsic<[], [llvm_ptr_ty]>; +def int_objc_retain : Intrinsic<[llvm_ptr_ty], + [llvm_ptr_ty], + [Returned>]>; + +def int_objc_retainAutorelease : Intrinsic<[llvm_ptr_ty], + [llvm_ptr_ty], + [Returned>]>; +def int_objc_retainAutoreleaseReturnValue : Intrinsic<[llvm_ptr_ty], + [llvm_ptr_ty], + [Returned>]>; def int_objc_retainAutoreleasedReturnValue : Intrinsic<[llvm_ptr_ty], [llvm_ptr_ty]>; def int_objc_unsafeClaimAutoreleasedReturnValue : Intrinsic<[llvm_ptr_ty], [llvm_ptr_ty]>; -def int_objc_claimAutoreleasedReturnValue : Intrinsic<[llvm_ptr_ty], - [llvm_ptr_ty]>; - -def int_objc_retainBlock : Intrinsic<[llvm_ptr_ty], - [llvm_ptr_ty]>; -def int_objc_storeStrong : Intrinsic<[], - [llvm_ptr_ty, - llvm_ptr_ty]>; -def int_objc_storeWeak : Intrinsic<[llvm_ptr_ty], - [llvm_ptr_ty, - llvm_ptr_ty]>; -def int_objc_clang_arc_use : Intrinsic<[], - [llvm_vararg_ty]>; -def int_objc_clang_arc_noop_use : DefaultAttrsIntrinsic<[], - [llvm_vararg_ty], - [IntrInaccessibleMemOnly]>; -def int_objc_retainedObject : Intrinsic<[llvm_ptr_ty], - [llvm_ptr_ty]>; -def int_objc_unretainedObject : Intrinsic<[llvm_ptr_ty], - [llvm_ptr_ty]>; -def int_objc_unretainedPointer : Intrinsic<[llvm_ptr_ty], - [llvm_ptr_ty]>; -def int_objc_retain_autorelease : Intrinsic<[llvm_ptr_ty], - [llvm_ptr_ty], - [Returned>]>; -def int_objc_sync_enter : Intrinsic<[llvm_i32_ty], - [llvm_ptr_ty]>; -def int_objc_sync_exit : Intrinsic<[llvm_i32_ty], - [llvm_ptr_ty]>; +def int_objc_claimAutoreleasedReturnValue : Intrinsic<[llvm_ptr_ty], + [llvm_ptr_ty]>; + +def int_objc_retainBlock : Intrinsic<[llvm_ptr_ty], + [llvm_ptr_ty]>; +def int_objc_storeStrong : Intrinsic<[], + [llvm_ptr_ty, + llvm_ptr_ty]>; +def int_objc_storeWeak : Intrinsic<[llvm_ptr_ty], + [llvm_ptr_ty, + llvm_ptr_ty]>; +def int_objc_clang_arc_use : Intrinsic<[], + [llvm_vararg_ty]>; +def int_objc_clang_arc_noop_use : DefaultAttrsIntrinsic<[], + [llvm_vararg_ty], + [IntrInaccessibleMemOnly]>; +def int_objc_retainedObject : Intrinsic<[llvm_ptr_ty], + [llvm_ptr_ty]>; +def int_objc_unretainedObject : Intrinsic<[llvm_ptr_ty], + [llvm_ptr_ty]>; +def int_objc_unretainedPointer : Intrinsic<[llvm_ptr_ty], + [llvm_ptr_ty]>; +def int_objc_retain_autorelease : Intrinsic<[llvm_ptr_ty], + [llvm_ptr_ty], + [Returned>]>; +def int_objc_sync_enter : Intrinsic<[llvm_i32_ty], + [llvm_ptr_ty]>; +def int_objc_sync_exit : Intrinsic<[llvm_i32_ty], + [llvm_ptr_ty]>; def int_objc_arc_annotation_topdown_bbstart : Intrinsic<[], [llvm_ptr_ty, llvm_ptr_ty]>; -def int_objc_arc_annotation_topdown_bbend : Intrinsic<[], - [llvm_ptr_ty, - llvm_ptr_ty]>; +def int_objc_arc_annotation_topdown_bbend : Intrinsic<[], + [llvm_ptr_ty, + llvm_ptr_ty]>; def int_objc_arc_annotation_bottomup_bbstart : Intrinsic<[], [llvm_ptr_ty, llvm_ptr_ty]>; @@ -904,15 +998,20 @@ def int_swift_async_context_addr : Intrinsic<[llvm_ptr_ty], [], []>; // def int_returnaddress : DefaultAttrsIntrinsic<[llvm_anyptr_ty], [llvm_i32_ty], [IntrNoMem, ImmArg>]>; -def int_addressofreturnaddress : DefaultAttrsIntrinsic<[llvm_anyptr_ty], [], [IntrNoMem]>; +def int_addressofreturnaddress : DefaultAttrsIntrinsic<[llvm_anyptr_ty], [], + [IntrNoMem]>; def int_frameaddress : DefaultAttrsIntrinsic<[llvm_anyptr_ty], [llvm_i32_ty], [IntrNoMem, ImmArg>]>; def int_sponentry : DefaultAttrsIntrinsic<[llvm_anyptr_ty], [], [IntrNoMem]>; def int_stackaddress : DefaultAttrsIntrinsic<[llvm_anyptr_ty], [], []>; -def int_read_register : DefaultAttrsIntrinsic<[llvm_anyint_ty], [llvm_metadata_ty], - [IntrReadMem], "llvm.read_register">; +def int_read_register : DefaultAttrsIntrinsic<[llvm_anyint_ty], + [llvm_metadata_ty], [IntrReadMem], + "llvm.read_register">; def int_write_register : Intrinsic<[], [llvm_metadata_ty, llvm_anyint_ty], [IntrNoCallback], "llvm.write_register">; +def int_write_volatile_register : Intrinsic<[], + [llvm_metadata_ty, llvm_anyint_ty], + [], "llvm.write_volatile_register">; def int_read_volatile_register : Intrinsic<[llvm_anyint_ty], [llvm_metadata_ty], [IntrHasSideEffects], "llvm.read_volatile_register">; @@ -955,7 +1054,8 @@ def int_stackrestore : DefaultAttrsIntrinsic<[], [llvm_anyptr_ty]>, def int_get_dynamic_area_offset : DefaultAttrsIntrinsic<[llvm_anyint_ty]>; -def int_thread_pointer : DefaultAttrsIntrinsic<[llvm_anyptr_ty], [], [IntrNoMem]>, +def int_thread_pointer : DefaultAttrsIntrinsic< + [llvm_anyptr_ty], [], [IntrNoMem]>, ClangBuiltin<"__builtin_thread_pointer">; // IntrInaccessibleMemOrArgMemOnly is a little more pessimistic than strictly @@ -963,10 +1063,14 @@ def int_thread_pointer : DefaultAttrsIntrinsic<[llvm_anyptr_ty], [], [IntrNoMem] // from being reordered overly much with respect to nearby access to the same // memory while not impeding optimization. def int_prefetch - : DefaultAttrsIntrinsic<[], [ llvm_anyptr_ty, llvm_i32_ty, llvm_i32_ty, llvm_i32_ty ], + : DefaultAttrsIntrinsic<[], + [llvm_anyptr_ty, llvm_i32_ty, llvm_i32_ty, llvm_i32_ty], [IntrInaccessibleMemOrArgMemOnly, ReadOnly>, NoCapture>, - ImmArg>, ImmArg>, ImmArg>]>; + ImmArg>, Range, 0, 2>, + ImmArg>, Range, 0, 4>, + ImmArg>, Range, 0, 2> + ]>; def int_pcmarker : DefaultAttrsIntrinsic<[], [llvm_i32_ty]>; def int_readcyclecounter : DefaultAttrsIntrinsic<[llvm_i64_ty]>; @@ -975,8 +1079,9 @@ def int_readsteadycounter : DefaultAttrsIntrinsic<[llvm_i64_ty]>; // The assume intrinsic is marked InaccessibleMemOnly so that proper control // dependencies will be maintained. -def int_assume : DefaultAttrsIntrinsic< - [], [llvm_i1_ty], [IntrWriteMem, IntrInaccessibleMemOnly, NoUndef>]>; +def int_assume : DefaultAttrsIntrinsic<[], [llvm_i1_ty], + [IntrWriteMem, IntrInaccessibleMemOnly, + NoUndef>]>; // 'llvm.experimental.noalias.scope.decl' intrinsic: Inserted at the location of // noalias scope declaration. Makes it possible to identify that a noalias scope @@ -990,8 +1095,8 @@ def int_experimental_noalias_scope_decl // Stack Protector Intrinsic - The stackprotector intrinsic writes the stack // guard to the correct place on the stack frame. -def int_stackprotector : DefaultAttrsIntrinsic<[], [llvm_ptr_ty, llvm_ptr_ty], []>; -def int_stackguard : DefaultAttrsIntrinsic<[llvm_ptr_ty], [], []>; +def int_stackprotector : DefaultAttrsIntrinsic<[], [llvm_ptr_ty, llvm_ptr_ty]>; +def int_stackguard : DefaultAttrsIntrinsic<[llvm_ptr_ty], []>; // A cover for instrumentation based profiling. def int_instrprof_cover : Intrinsic<[], [llvm_ptr_ty, llvm_i64_ty, @@ -1004,13 +1109,13 @@ def int_instrprof_increment : Intrinsic<[], // A counter increment with step for instrumentation based profiling. def int_instrprof_increment_step : Intrinsic<[], - [llvm_ptr_ty, llvm_i64_ty, - llvm_i32_ty, llvm_i32_ty, llvm_i64_ty]>; + [llvm_ptr_ty, llvm_i64_ty, + llvm_i32_ty, llvm_i32_ty, llvm_i64_ty]>; // Callsite instrumentation for contextual profiling def int_instrprof_callsite : Intrinsic<[], - [llvm_ptr_ty, llvm_i64_ty, - llvm_i32_ty, llvm_i32_ty, llvm_ptr_ty]>; + [llvm_ptr_ty, llvm_i64_ty, + llvm_i32_ty, llvm_i32_ty, llvm_ptr_ty]>; // A timestamp for instrumentation based profiling. def int_instrprof_timestamp : Intrinsic<[], [llvm_ptr_ty, llvm_i64_ty, @@ -1051,7 +1156,8 @@ def int_structured_gep [LLVMMatchType<0>, llvm_vararg_ty], [IntrNoMem, IntrSpeculatable]>; -def int_structured_alloca : DefaultAttrsIntrinsic<[llvm_anyptr_ty], [], [IntrInaccessibleMemOnly]>; +def int_structured_alloca : DefaultAttrsIntrinsic<[llvm_anyptr_ty], [], + [IntrInaccessibleMemOnly]>; //===------------------- Standard C Library Intrinsics --------------------===// // @@ -1098,7 +1204,6 @@ def int_memset : DefaultAttrsIntrinsic<[], // Memset version that is guaranteed to be inlined. // In particular this means that the generated code is not allowed to call any // external function. -// The third argument (specifying the size) must be a constant. def int_memset_inline : DefaultAttrsIntrinsic<[], [llvm_anyptr_ty, llvm_i8_ty, llvm_anyint_ty, llvm_i1_ty], @@ -1120,7 +1225,8 @@ def int_experimental_memset_pattern // FIXME: Add version of these floating point intrinsics which allow non-default // rounding modes and FP exception handling. -let IntrProperties = [IntrNoMem, IntrSpeculatable, IntrNoCreateUndefOrPoison] in { +let IntrProperties = [IntrNoMem, IntrSpeculatable, IntrNoCreateUndefOrPoison] + in { def int_fma : DefaultAttrsIntrinsic<[llvm_anyfloat_ty], [LLVMMatchType<0>, LLVMMatchType<0>, LLVMMatchType<0>]>; @@ -1132,7 +1238,9 @@ let IntrProperties = [IntrNoMem, IntrSpeculatable, IntrNoCreateUndefOrPoison] in // rounding mode. LLVM purposely does not model changes to the FP // environment so they can be treated as readnone. def int_sqrt : DefaultAttrsIntrinsic<[llvm_anyfloat_ty], [LLVMMatchType<0>]>; - def int_powi : DefaultAttrsIntrinsic<[llvm_anyfloat_ty], [LLVMMatchType<0>, llvm_anyint_ty]>; + def int_powi : DefaultAttrsIntrinsic< + [llvm_anyfloat_ty], + [LLVMMatchType<0>, llvm_any_scalar_int_ty]>; def int_sin : DefaultAttrsIntrinsic<[llvm_anyfloat_ty], [LLVMMatchType<0>]>; def int_cos : DefaultAttrsIntrinsic<[llvm_anyfloat_ty], [LLVMMatchType<0>]>; def int_pow : DefaultAttrsIntrinsic<[llvm_anyfloat_ty], @@ -1150,13 +1258,15 @@ let IntrProperties = [IntrNoMem, IntrSpeculatable, IntrNoCreateUndefOrPoison] in def int_ceil : DefaultAttrsIntrinsic<[llvm_anyfloat_ty], [LLVMMatchType<0>]>; def int_trunc : DefaultAttrsIntrinsic<[llvm_anyfloat_ty], [LLVMMatchType<0>]>; def int_rint : DefaultAttrsIntrinsic<[llvm_anyfloat_ty], [LLVMMatchType<0>]>; - def int_nearbyint : DefaultAttrsIntrinsic<[llvm_anyfloat_ty], [LLVMMatchType<0>]>; + def int_nearbyint : DefaultAttrsIntrinsic< + [llvm_anyfloat_ty], [LLVMMatchType<0>]>; def int_round : DefaultAttrsIntrinsic<[llvm_anyfloat_ty], [LLVMMatchType<0>]>; - def int_roundeven : DefaultAttrsIntrinsic<[llvm_anyfloat_ty], [LLVMMatchType<0>]>; + def int_roundeven : DefaultAttrsIntrinsic< + [llvm_anyfloat_ty], [LLVMMatchType<0>]>; // Truncate a floating point number with a specific rounding mode def int_fptrunc_round : DefaultAttrsIntrinsic<[ llvm_anyfloat_ty ], - [ llvm_anyfloat_ty, llvm_metadata_ty ]>; + [ llvm_anyfloat_ty, llvm_metadata_ty ]>; // Convert from native LLVM floating-point to arbitrary FP format // Returns an integer containing the arbitrary FP bits @@ -1174,11 +1284,13 @@ let IntrProperties = [IntrNoMem, IntrSpeculatable, IntrNoCreateUndefOrPoison] in [ llvm_anyint_ty, llvm_metadata_ty ], [ IntrNoMem, IntrSpeculatable ]>; - def int_canonicalize : DefaultAttrsIntrinsic<[llvm_anyfloat_ty], [LLVMMatchType<0>], - [IntrNoMem]>; + def int_canonicalize : DefaultAttrsIntrinsic< + [llvm_anyfloat_ty], [LLVMMatchType<0>], + [IntrNoMem]>; // Arithmetic fence intrinsic. - def int_arithmetic_fence : DefaultAttrsIntrinsic<[llvm_anyfloat_ty], [LLVMMatchType<0>], - [IntrNoMem]>; + def int_arithmetic_fence : DefaultAttrsIntrinsic< + [llvm_anyfloat_ty], [LLVMMatchType<0>], + [IntrNoMem]>; // If the value doesn't fit an unspecified value is returned, but this // is not poison so we can still mark these as IntrNoCreateUndefOrPoison. @@ -1187,12 +1299,14 @@ let IntrProperties = [IntrNoMem, IntrSpeculatable, IntrNoCreateUndefOrPoison] in def int_lrint : DefaultAttrsIntrinsic<[llvm_anyint_ty], [llvm_anyfloat_ty]>; def int_llrint : DefaultAttrsIntrinsic<[llvm_anyint_ty], [llvm_anyfloat_ty]>; - // TODO: int operand should be constrained to same number of elements as the result. + // TODO: int operand should be constrained to same number of elements as the + // result. def int_ldexp : DefaultAttrsIntrinsic<[llvm_anyfloat_ty], [LLVMMatchType<0>, llvm_anyint_ty]>; - // TODO: Should constrain all element counts to match - def int_frexp : DefaultAttrsIntrinsic<[llvm_anyfloat_ty, llvm_anyint_ty], [LLVMMatchType<0>]>; + // TODO: Should constrain all element counts to match. + def int_frexp : DefaultAttrsIntrinsic<[llvm_anyfloat_ty, llvm_anyint_ty], + [LLVMMatchType<0>]>; } // TODO: Move all of these into the IntrNoCreateUndefOrPoison case above. @@ -1203,7 +1317,8 @@ let IntrProperties = [IntrNoMem, IntrSpeculatable] in { def int_asin : DefaultAttrsIntrinsic<[llvm_anyfloat_ty], [LLVMMatchType<0>]>; def int_acos : DefaultAttrsIntrinsic<[llvm_anyfloat_ty], [LLVMMatchType<0>]>; def int_atan : DefaultAttrsIntrinsic<[llvm_anyfloat_ty], [LLVMMatchType<0>]>; - def int_atan2 : DefaultAttrsIntrinsic<[llvm_anyfloat_ty], [LLVMMatchType<0>, LLVMMatchType<0>]>; + def int_atan2 : DefaultAttrsIntrinsic< + [llvm_anyfloat_ty], [LLVMMatchType<0>, LLVMMatchType<0>]>; def int_tan : DefaultAttrsIntrinsic<[llvm_anyfloat_ty], [LLVMMatchType<0>]>; def int_sinh : DefaultAttrsIntrinsic<[llvm_anyfloat_ty], [LLVMMatchType<0>]>; def int_cosh : DefaultAttrsIntrinsic<[llvm_anyfloat_ty], [LLVMMatchType<0>]>; @@ -1270,7 +1385,10 @@ let IntrProperties = [IntrInaccessibleMemOnly] in { def int_is_fpclass : DefaultAttrsIntrinsic<[LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>], [llvm_anyfloat_ty, llvm_i32_ty], - [IntrNoMem, IntrSpeculatable, IntrNoCreateUndefOrPoison, ImmArg>]>; + [IntrNoMem, IntrSpeculatable, + IntrNoCreateUndefOrPoison, ImmArg>, + ArgInfo, + [ImmArgPrinter<"printFPClassMask">]>]>; //===--------------- Constrained Floating Point Intrinsics ----------------===// // @@ -1439,12 +1557,12 @@ let IntrProperties = [IntrInaccessibleMemOnly, IntrStrictFP] in { [ LLVMMatchType<0>, llvm_metadata_ty, llvm_metadata_ty ]>; - def int_experimental_constrained_lrint : DefaultAttrsIntrinsic<[ llvm_anyint_ty ], - [ llvm_anyfloat_ty, + def int_experimental_constrained_lrint : DefaultAttrsIntrinsic<[ llvm_any_scalar_int_ty ], + [ llvm_any_scalar_float_ty, llvm_metadata_ty, llvm_metadata_ty ]>; - def int_experimental_constrained_llrint : DefaultAttrsIntrinsic<[ llvm_anyint_ty ], - [ llvm_anyfloat_ty, + def int_experimental_constrained_llrint : DefaultAttrsIntrinsic<[ llvm_any_scalar_int_ty ], + [ llvm_any_scalar_float_ty, llvm_metadata_ty, llvm_metadata_ty ]>; def int_experimental_constrained_maxnum : DefaultAttrsIntrinsic<[ llvm_anyfloat_ty ], @@ -1469,11 +1587,11 @@ let IntrProperties = [IntrInaccessibleMemOnly, IntrStrictFP] in { def int_experimental_constrained_floor : DefaultAttrsIntrinsic<[ llvm_anyfloat_ty ], [ LLVMMatchType<0>, llvm_metadata_ty ]>; - def int_experimental_constrained_lround : DefaultAttrsIntrinsic<[ llvm_anyint_ty ], - [ llvm_anyfloat_ty, + def int_experimental_constrained_lround : DefaultAttrsIntrinsic<[ llvm_any_scalar_int_ty ], + [ llvm_any_scalar_float_ty, llvm_metadata_ty ]>; - def int_experimental_constrained_llround : DefaultAttrsIntrinsic<[ llvm_anyint_ty ], - [ llvm_anyfloat_ty, + def int_experimental_constrained_llround : DefaultAttrsIntrinsic<[ llvm_any_scalar_int_ty ], + [ llvm_any_scalar_float_ty, llvm_metadata_ty ]>; def int_experimental_constrained_round : DefaultAttrsIntrinsic<[ llvm_anyfloat_ty ], [ LLVMMatchType<0>, @@ -1512,7 +1630,8 @@ def int_expect_with_probability : DefaultAttrsIntrinsic<[llvm_anyint_ty], // // None of these intrinsics accesses memory at all. -let IntrProperties = [IntrNoMem, IntrSpeculatable, IntrNoCreateUndefOrPoison] in { +let IntrProperties = [IntrNoMem, IntrSpeculatable, IntrNoCreateUndefOrPoison] + in { def int_bswap: DefaultAttrsIntrinsic<[llvm_anyint_ty], [LLVMMatchType<0>]>; def int_ctpop: DefaultAttrsIntrinsic<[llvm_anyint_ty], [LLVMMatchType<0>]>; def int_bitreverse : DefaultAttrsIntrinsic<[llvm_anyint_ty], [LLVMMatchType<0>]>; @@ -1522,12 +1641,17 @@ let IntrProperties = [IntrNoMem, IntrSpeculatable, IntrNoCreateUndefOrPoison] in [LLVMMatchType<0>, LLVMMatchType<0>, LLVMMatchType<0>]>; def int_clmul : DefaultAttrsIntrinsic<[llvm_anyint_ty], [LLVMMatchType<0>, LLVMMatchType<0>]>; + def int_pext : DefaultAttrsIntrinsic<[llvm_anyint_ty], + [LLVMMatchType<0>, LLVMMatchType<0>]>; + def int_pdep : DefaultAttrsIntrinsic<[llvm_anyint_ty], + [LLVMMatchType<0>, LLVMMatchType<0>]>; } -let IntrProperties = [IntrNoMem, IntrSpeculatable, - ImmArg>] in { - def int_ctlz : DefaultAttrsIntrinsic<[llvm_anyint_ty], [LLVMMatchType<0>, llvm_i1_ty]>; - def int_cttz : DefaultAttrsIntrinsic<[llvm_anyint_ty], [LLVMMatchType<0>, llvm_i1_ty]>; +let IntrProperties = [IntrNoMem, IntrSpeculatable, ImmArg>] in { + def int_ctlz : DefaultAttrsIntrinsic<[llvm_anyint_ty], + [LLVMMatchType<0>, llvm_i1_ty]>; + def int_cttz : DefaultAttrsIntrinsic<[llvm_anyint_ty], + [LLVMMatchType<0>, llvm_i1_ty]>; } //===------------------------ Debugger Intrinsics -------------------------===// @@ -1584,7 +1708,8 @@ def int_eh_unwind_init: Intrinsic<[]>, def int_eh_dwarf_cfa : Intrinsic<[llvm_ptr_ty], [llvm_i32_ty]>; def int_eh_sjlj_lsda : Intrinsic<[llvm_ptr_ty], [], [IntrNoMem]>; -def int_eh_sjlj_callsite : Intrinsic<[], [llvm_i32_ty], [IntrNoMem, ImmArg>]>; +def int_eh_sjlj_callsite : Intrinsic<[], [llvm_i32_ty], + [IntrNoMem, ImmArg>]>; def int_eh_sjlj_functioncontext : Intrinsic<[], [llvm_ptr_ty]>; def int_eh_sjlj_setjmp : Intrinsic<[llvm_i32_ty], [llvm_ptr_ty]>; @@ -1594,12 +1719,15 @@ def int_eh_sjlj_setup_dispatch : Intrinsic<[], []>; //===---------------- Generic Variable Attribute Intrinsics----------------===// // def int_var_annotation : DefaultAttrsIntrinsic< - [], [llvm_anyptr_ty, llvm_anyptr_ty, LLVMMatchType<1>, llvm_i32_ty, LLVMMatchType<1>], + [], + [llvm_anyptr_ty, llvm_anyptr_ty, LLVMMatchType<1>, llvm_i32_ty, + LLVMMatchType<1>], [IntrInaccessibleMemOnly]>; def int_ptr_annotation : DefaultAttrsIntrinsic< [llvm_anyptr_ty], - [LLVMMatchType<0>, llvm_anyptr_ty, LLVMMatchType<1>, llvm_i32_ty, LLVMMatchType<1>], + [LLVMMatchType<0>, llvm_anyptr_ty, LLVMMatchType<1>, llvm_i32_ty, + LLVMMatchType<1>], [IntrInaccessibleMemOnly]>; def int_annotation : DefaultAttrsIntrinsic< @@ -1611,7 +1739,7 @@ def int_annotation : DefaultAttrsIntrinsic< // as CodeView debug info records. This is expensive, as it disables inlining // and is modelled as having side effects. def int_codeview_annotation : DefaultAttrsIntrinsic<[], [llvm_metadata_ty], - [IntrInaccessibleMemOnly, IntrNoDuplicate]>; + [IntrInaccessibleMemOnly, IntrNoDuplicate]>; //===------------------------ Trampoline Intrinsics -----------------------===// // @@ -1629,42 +1757,53 @@ def int_adjust_trampoline : DefaultAttrsIntrinsic< // // Expose the carry flag from add operations on two integrals. -let IntrProperties = [IntrNoMem, IntrSpeculatable, IntrNoCreateUndefOrPoison] in { - def int_sadd_with_overflow : DefaultAttrsIntrinsic<[llvm_anyint_ty, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>], - [LLVMMatchType<0>, LLVMMatchType<0>]>; - def int_uadd_with_overflow : DefaultAttrsIntrinsic<[llvm_anyint_ty, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>], - [LLVMMatchType<0>, LLVMMatchType<0>]>; - - def int_ssub_with_overflow : DefaultAttrsIntrinsic<[llvm_anyint_ty, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>], - [LLVMMatchType<0>, LLVMMatchType<0>]>; - def int_usub_with_overflow : DefaultAttrsIntrinsic<[llvm_anyint_ty, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>], - [LLVMMatchType<0>, LLVMMatchType<0>]>; - - def int_smul_with_overflow : DefaultAttrsIntrinsic<[llvm_anyint_ty, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>], - [LLVMMatchType<0>, LLVMMatchType<0>]>; - def int_umul_with_overflow : DefaultAttrsIntrinsic<[llvm_anyint_ty, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>], - [LLVMMatchType<0>, LLVMMatchType<0>]>; +let IntrProperties = [IntrNoMem, IntrSpeculatable, IntrNoCreateUndefOrPoison] + in { + def int_sadd_with_overflow : DefaultAttrsIntrinsic< + [llvm_anyint_ty, + LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>], + [LLVMMatchType<0>, LLVMMatchType<0>]>; + def int_uadd_with_overflow : DefaultAttrsIntrinsic< + [llvm_anyint_ty, + LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>], + [LLVMMatchType<0>, LLVMMatchType<0>]>; + + def int_ssub_with_overflow : DefaultAttrsIntrinsic< + [llvm_anyint_ty, + LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>], + [LLVMMatchType<0>, LLVMMatchType<0>]>; + def int_usub_with_overflow : DefaultAttrsIntrinsic< + [llvm_anyint_ty, + LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>], + [LLVMMatchType<0>, LLVMMatchType<0>]>; + + def int_smul_with_overflow : DefaultAttrsIntrinsic< + [llvm_anyint_ty, + LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>], + [LLVMMatchType<0>, LLVMMatchType<0>]>; + def int_umul_with_overflow : DefaultAttrsIntrinsic< + [llvm_anyint_ty, + LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>], + [LLVMMatchType<0>, LLVMMatchType<0>]>; } -//===------------------------- Saturation Arithmetic Intrinsics ---------------------===// +//===------------------ Saturation Arithmetic Intrinsics ------------------===// // def int_sadd_sat : DefaultAttrsIntrinsic<[llvm_anyint_ty], [LLVMMatchType<0>, LLVMMatchType<0>], - [IntrNoMem, IntrSpeculatable, IntrNoCreateUndefOrPoison, Commutative]>; + [IntrNoMem, IntrSpeculatable, + IntrNoCreateUndefOrPoison, Commutative]>; def int_uadd_sat : DefaultAttrsIntrinsic<[llvm_anyint_ty], [LLVMMatchType<0>, LLVMMatchType<0>], - [IntrNoMem, IntrSpeculatable, IntrNoCreateUndefOrPoison, Commutative]>; + [IntrNoMem, IntrSpeculatable, + IntrNoCreateUndefOrPoison, Commutative]>; def int_ssub_sat : DefaultAttrsIntrinsic<[llvm_anyint_ty], [LLVMMatchType<0>, LLVMMatchType<0>], - [IntrNoMem, IntrSpeculatable, IntrNoCreateUndefOrPoison]>; + [IntrNoMem, IntrSpeculatable, + IntrNoCreateUndefOrPoison]>; def int_usub_sat : DefaultAttrsIntrinsic<[llvm_anyint_ty], [LLVMMatchType<0>, LLVMMatchType<0>], - [IntrNoMem, IntrSpeculatable, IntrNoCreateUndefOrPoison]>; + [IntrNoMem, IntrSpeculatable, + IntrNoCreateUndefOrPoison]>; def int_sshl_sat : DefaultAttrsIntrinsic<[llvm_anyint_ty], [LLVMMatchType<0>, LLVMMatchType<0>], [IntrNoMem, IntrSpeculatable]>; @@ -1672,7 +1811,7 @@ def int_ushl_sat : DefaultAttrsIntrinsic<[llvm_anyint_ty], [LLVMMatchType<0>, LLVMMatchType<0>], [IntrNoMem, IntrSpeculatable]>; -//===------------------------- Fixed Point Arithmetic Intrinsics ---------------------===// +//===------------------- Fixed Point Arithmetic Intrinsics ----------------===// // def int_smul_fix : DefaultAttrsIntrinsic<[llvm_anyint_ty], [LLVMMatchType<0>, LLVMMatchType<0>, llvm_i32_ty], @@ -1692,24 +1831,24 @@ def int_udiv_fix : DefaultAttrsIntrinsic<[llvm_anyint_ty], [LLVMMatchType<0>, LLVMMatchType<0>, llvm_i32_ty], [IntrNoMem, ImmArg>]>; -//===------------------- Fixed Point Saturation Arithmetic Intrinsics ----------------===// +//===------------ Fixed Point Saturation Arithmetic Intrinsics ------------===// // def int_smul_fix_sat : DefaultAttrsIntrinsic<[llvm_anyint_ty], - [LLVMMatchType<0>, LLVMMatchType<0>, llvm_i32_ty], - [IntrNoMem, IntrSpeculatable, - Commutative, ImmArg>]>; + [LLVMMatchType<0>, LLVMMatchType<0>, llvm_i32_ty], + [IntrNoMem, IntrSpeculatable, + Commutative, ImmArg>]>; def int_umul_fix_sat : DefaultAttrsIntrinsic<[llvm_anyint_ty], - [LLVMMatchType<0>, LLVMMatchType<0>, llvm_i32_ty], - [IntrNoMem, IntrSpeculatable, - Commutative, ImmArg>]>; + [LLVMMatchType<0>, LLVMMatchType<0>, llvm_i32_ty], + [IntrNoMem, IntrSpeculatable, + Commutative, ImmArg>]>; def int_sdiv_fix_sat : DefaultAttrsIntrinsic<[llvm_anyint_ty], - [LLVMMatchType<0>, LLVMMatchType<0>, llvm_i32_ty], - [IntrNoMem, ImmArg>]>; + [LLVMMatchType<0>, LLVMMatchType<0>, llvm_i32_ty], + [IntrNoMem, ImmArg>]>; def int_udiv_fix_sat : DefaultAttrsIntrinsic<[llvm_anyint_ty], - [LLVMMatchType<0>, LLVMMatchType<0>, llvm_i32_ty], - [IntrNoMem, ImmArg>]>; + [LLVMMatchType<0>, LLVMMatchType<0>, llvm_i32_ty], + [IntrNoMem, ImmArg>]>; //===------------------ Integer Min/Max/Abs Intrinsics --------------------===// // @@ -1731,10 +1870,12 @@ def int_umin : DefaultAttrsIntrinsic< [IntrNoMem, IntrSpeculatable, IntrNoCreateUndefOrPoison]>; def int_scmp : DefaultAttrsIntrinsic< [llvm_anyint_ty], [llvm_anyint_ty, LLVMMatchType<1>], - [IntrNoMem, IntrSpeculatable, IntrNoCreateUndefOrPoison, Range]>; + [IntrNoMem, IntrSpeculatable, IntrNoCreateUndefOrPoison, + Range]>; def int_ucmp : DefaultAttrsIntrinsic< [llvm_anyint_ty], [llvm_anyint_ty, LLVMMatchType<1>], - [IntrNoMem, IntrSpeculatable, IntrNoCreateUndefOrPoison, Range]>; + [IntrNoMem, IntrSpeculatable, IntrNoCreateUndefOrPoison, + Range]>; //===------------------------- Memory Use Markers -------------------------===// // @@ -1769,8 +1910,9 @@ def int_invariant_end : DefaultAttrsIntrinsic<[], // Note that it is still experimental, which means that its semantics // might change in the future. def int_launder_invariant_group : DefaultAttrsIntrinsic<[llvm_anyptr_ty], - [LLVMMatchType<0>], - [IntrInaccessibleMemOnly, IntrSpeculatable]>; + [LLVMMatchType<0>], + [IntrInaccessibleMemOnly, + IntrSpeculatable]>; def int_strip_invariant_group : DefaultAttrsIntrinsic<[llvm_anyptr_ty], @@ -1781,7 +1923,8 @@ def int_strip_invariant_group : DefaultAttrsIntrinsic<[llvm_anyptr_ty], // def int_experimental_stackmap : DefaultAttrsIntrinsic<[], [llvm_i64_ty, llvm_i32_ty, llvm_vararg_ty], - [Throws, ImmArg>, ImmArg>]>; + [Throws, ImmArg>, + ImmArg>]>; def int_experimental_patchpoint_void : Intrinsic<[], [llvm_i64_ty, llvm_i32_ty, llvm_ptr_ty, llvm_i32_ty, @@ -1817,7 +1960,7 @@ def int_experimental_gc_relocate : DefaultAttrsIntrinsic< [IntrNoMem, ImmArg>, ImmArg>]>; def int_experimental_gc_get_pointer_base : DefaultAttrsIntrinsic< - [llvm_anyptr_ty], [llvm_anyptr_ty], + [llvm_anyptr_ty], [LLVMMatchType<0>], [IntrNoMem, ReadNone>, NoCapture>]>; def int_experimental_gc_get_pointer_offset : DefaultAttrsIntrinsic< @@ -1825,7 +1968,7 @@ def int_experimental_gc_get_pointer_offset : DefaultAttrsIntrinsic< [IntrNoMem, ReadNone>, NoCapture>]>; //===------------------------ Coroutine Intrinsics ---------------===// -// These are documented in docs/Coroutines.rst +// These are documented in docs/Coroutines.md // Coroutine Structure Intrinsics. @@ -1841,7 +1984,7 @@ def int_coro_id_retcon_once : Intrinsic<[llvm_token_ty], [llvm_i32_ty, llvm_i32_ty, llvm_ptr_ty, llvm_ptr_ty, llvm_ptr_ty, llvm_ptr_ty], []>; -def int_coro_alloc : Intrinsic<[llvm_i1_ty], [llvm_token_ty], []>; +def int_coro_alloc : Intrinsic<[llvm_i1_ty], [llvm_token_ty], [IntrNoMem]>; def int_coro_id_async : Intrinsic<[llvm_token_ty], [llvm_i32_ty, llvm_i32_ty, llvm_i32_ty, llvm_ptr_ty], []>; @@ -1862,9 +2005,10 @@ def int_coro_suspend_async def int_coro_prepare_async : Intrinsic<[llvm_ptr_ty], [llvm_ptr_ty], [IntrNoMem]>; def int_coro_begin : Intrinsic<[llvm_ptr_ty], [llvm_token_ty, llvm_ptr_ty], - [WriteOnly>]>; -def int_coro_begin_custom_abi : Intrinsic<[llvm_ptr_ty], [llvm_token_ty, llvm_ptr_ty, llvm_i32_ty], - [WriteOnly>]>; + [IntrArgMemOnly, WriteOnly>]>; +def int_coro_begin_custom_abi : Intrinsic<[llvm_ptr_ty], + [llvm_token_ty, llvm_ptr_ty, llvm_i32_ty], + [IntrArgMemOnly, WriteOnly>]>; def int_coro_free : Intrinsic<[llvm_ptr_ty], [llvm_token_ty, llvm_ptr_ty], [IntrReadMem, IntrArgMemOnly, ReadOnly>, @@ -1876,7 +2020,8 @@ def int_coro_end_async : Intrinsic<[], [llvm_ptr_ty, llvm_i1_ty, llvm_vararg_ty], []>; def int_coro_frame : Intrinsic<[llvm_ptr_ty], [], [IntrNoMem]>; -def int_coro_is_in_ramp : Intrinsic<[llvm_i1_ty], [], [IntrNoMem], "llvm.coro.is_in_ramp">; +def int_coro_is_in_ramp : Intrinsic<[llvm_i1_ty], [], [IntrNoMem], + "llvm.coro.is_in_ramp">; def int_coro_noop : Intrinsic<[llvm_ptr_ty], [], [IntrNoMem]>; def int_coro_size : Intrinsic<[llvm_anyint_ty], [], [IntrNoMem]>; def int_coro_align : Intrinsic<[llvm_anyint_ty], [], [IntrNoMem]>; @@ -1907,12 +2052,12 @@ def int_coro_await_suspend_void : Intrinsic<[], [Throws]>; def int_coro_await_suspend_bool : Intrinsic<[llvm_i1_ty], - [llvm_ptr_ty, llvm_ptr_ty, llvm_ptr_ty], - [Throws]>; + [llvm_ptr_ty, llvm_ptr_ty, llvm_ptr_ty], + [Throws]>; def int_coro_await_suspend_handle : Intrinsic<[], - [llvm_ptr_ty, llvm_ptr_ty, llvm_ptr_ty], - [Throws]>; + [llvm_ptr_ty, llvm_ptr_ty, llvm_ptr_ty], + [Throws]>; // Coroutine Lowering Intrinsics. Used internally by coroutine passes. @@ -1921,12 +2066,14 @@ def int_coro_subfn_addr : DefaultAttrsIntrinsic< [IntrReadMem, IntrArgMemOnly, ReadOnly>, NoCapture>]>; -///===-------------------------- Other Intrinsics --------------------------===// +///===------------------------- Other Intrinsics --------------------------===// // // TODO: We should introduce a new memory kind fo traps (and other side effects // we only model to keep things alive). -def int_trap : Intrinsic<[], [], [IntrNoReturn, IntrCold, IntrInaccessibleMemOnly, - IntrWriteMem]>, ClangBuiltin<"__builtin_trap">; +def int_trap : Intrinsic<[], [], + [IntrNoReturn, IntrCold, IntrInaccessibleMemOnly, + IntrWriteMem]>, + ClangBuiltin<"__builtin_trap">; def int_debugtrap : Intrinsic<[]>, ClangBuiltin<"__builtin_debugtrap">; def int_ubsantrap : Intrinsic<[], [llvm_i8_ty], @@ -1940,8 +2087,8 @@ def int_allow_ubsan_check : DefaultAttrsIntrinsic<[llvm_i1_ty], [llvm_i8_ty], [IntrInaccessibleMemOnly, ImmArg>, NoUndef]>; // Return true if runtime check is allowed. -def int_allow_runtime_check : DefaultAttrsIntrinsic<[llvm_i1_ty], [llvm_metadata_ty], - [IntrInaccessibleMemOnly, NoUndef]>, +def int_allow_runtime_check : DefaultAttrsIntrinsic<[llvm_i1_ty], + [llvm_metadata_ty], [IntrInaccessibleMemOnly, NoUndef]>, ClangBuiltin<"__builtin_allow_runtime_check">; // Return true if the specific sanitizer is enabled for the function. @@ -1967,8 +2114,8 @@ def int_experimental_guard : Intrinsic<[], [llvm_i1_ty, llvm_vararg_ty], [Throws]>; // Supports widenable conditions for guards represented as explicit branches. -def int_experimental_widenable_condition : DefaultAttrsIntrinsic<[llvm_i1_ty], [], - [IntrInaccessibleMemOnly, IntrSpeculatable, NoUndef]>; +def int_experimental_widenable_condition : DefaultAttrsIntrinsic<[llvm_i1_ty], + [], [IntrInaccessibleMemOnly, IntrSpeculatable, NoUndef]>; // NOP: calls/invokes to this intrinsic are removed by codegen def int_donothing : DefaultAttrsIntrinsic<[], [], [IntrNoMem]>; @@ -1983,13 +2130,17 @@ def int_sideeffect : DefaultAttrsIntrinsic<[], [], [IntrInaccessibleMemOnly]>; // Like the sideeffect intrinsic defined above, this intrinsic is treated by the // optimizer as having opaque side effects so that it won't be get rid of or moved // out of the block it probes. -def int_pseudoprobe : DefaultAttrsIntrinsic<[], [llvm_i64_ty, llvm_i64_ty, llvm_i32_ty, llvm_i64_ty], - [IntrInaccessibleMemOnly]>; +def int_pseudoprobe : DefaultAttrsIntrinsic<[], + [llvm_i64_ty, llvm_i64_ty, llvm_i32_ty, llvm_i64_ty], + [IntrInaccessibleMemOnly]>; // Saturating floating point to integer intrinsics -let IntrProperties = [IntrNoMem, IntrSpeculatable, IntrNoCreateUndefOrPoison] in { -def int_fptoui_sat : DefaultAttrsIntrinsic<[llvm_anyint_ty], [llvm_anyfloat_ty]>; -def int_fptosi_sat : DefaultAttrsIntrinsic<[llvm_anyint_ty], [llvm_anyfloat_ty]>; +let IntrProperties = [IntrNoMem, IntrSpeculatable, IntrNoCreateUndefOrPoison] + in { +def int_fptoui_sat : DefaultAttrsIntrinsic<[llvm_anyint_ty], + [llvm_anyfloat_ty]>; +def int_fptosi_sat : DefaultAttrsIntrinsic<[llvm_anyint_ty], + [llvm_anyfloat_ty]>; } // Clear cache intrinsic, default to ignore (ie. emit nothing) @@ -2011,9 +2162,10 @@ def int_fake_use : DefaultAttrsIntrinsic<[], [llvm_vararg_ty], def int_ptrmask: PureIntrinsic<[llvm_any_ty], [LLVMMatchType<0>, llvm_anyint_ty]>; // Intrinsic to wrap a thread local variable. -def int_threadlocal_address : DefaultAttrsIntrinsic<[llvm_anyptr_ty], [LLVMMatchType<0>], - [NonNull, NonNull>, - IntrNoMem, IntrSpeculatable]>; +def int_threadlocal_address : DefaultAttrsIntrinsic<[llvm_anyptr_ty], + [LLVMMatchType<0>], + [NonNull, NonNull>, + IntrNoMem, IntrSpeculatable]>; def int_stepvector : DefaultAttrsIntrinsic<[llvm_anyvector_ty], [], [IntrNoMem]>; @@ -2021,26 +2173,29 @@ def int_stepvector : DefaultAttrsIntrinsic<[llvm_anyvector_ty], def int_reloc_none : DefaultAttrsIntrinsic<[], [llvm_metadata_ty], [IntrNoMem, IntrHasSideEffects]>; -//===---------------- Vector Predication Intrinsics --------------===// +//===--------------------- Vector Predication Intrinsics ------------------===// // Memory Intrinsics def int_vp_store : DefaultAttrsIntrinsic<[], [ llvm_anyvector_ty, llvm_anyptr_ty, LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, llvm_i32_ty], - [ NoCapture>, IntrWriteMem, IntrArgMemOnly ]>; + [ NoCapture>, IntrWriteMem, + IntrArgMemOnly ]>; def int_vp_load : DefaultAttrsIntrinsic<[ llvm_anyvector_ty], [ llvm_anyptr_ty, LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, llvm_i32_ty], - [ NoCapture>, IntrReadMem, IntrArgMemOnly ]>; + [ NoCapture>, IntrReadMem, + IntrArgMemOnly ]>; def int_vp_load_ff : DefaultAttrsIntrinsic<[ llvm_anyvector_ty, llvm_i32_ty ], [ llvm_anyptr_ty, LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, llvm_i32_ty], - [ NoCapture>, IntrNoSync, IntrReadMem, IntrWillReturn, IntrArgMemOnly ]>; + [ NoCapture>, IntrNoSync, IntrReadMem, + IntrWillReturn, IntrArgMemOnly ]>; def int_vp_gather: DefaultAttrsIntrinsic<[ llvm_anyvector_ty], [ LLVMVectorOfAnyPointersToElt<0>, @@ -2053,7 +2208,8 @@ def int_vp_scatter: DefaultAttrsIntrinsic<[], LLVMVectorOfAnyPointersToElt<0>, LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, llvm_i32_ty], - []>; // TODO allow IntrNoCapture for vectors of pointers + []>; // TODO allow IntrNoCapture for vectors of + // pointers. // Experimental strided memory accesses def int_experimental_vp_strided_store : DefaultAttrsIntrinsic<[], @@ -2062,47 +2218,49 @@ def int_experimental_vp_strided_store : DefaultAttrsIntrinsic<[], llvm_anyint_ty, // Stride in bytes LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, llvm_i32_ty], - [ NoCapture>, IntrWriteMem, IntrArgMemOnly ]>; + [ NoCapture>, IntrWriteMem, + IntrArgMemOnly ]>; def int_experimental_vp_strided_load : DefaultAttrsIntrinsic<[llvm_anyvector_ty], [ llvm_anyptr_ty, llvm_anyint_ty, // Stride in bytes LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, llvm_i32_ty], - [ NoCapture>, IntrReadMem, IntrArgMemOnly ]>; + [ NoCapture>, IntrReadMem, + IntrArgMemOnly ]>; // Experimental histogram def int_experimental_vector_histogram_add : DefaultAttrsIntrinsic<[], - [ llvm_anyvector_ty, // Vector of pointers - llvm_anyint_ty, // Increment - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>], // Mask - [ IntrArgMemOnly ]>; + [ llvm_anyvector_ty, // Vector of pointers + llvm_anyint_ty, // Increment + LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>], // Mask + [ IntrArgMemOnly ]>; def int_experimental_vector_histogram_uadd_sat : DefaultAttrsIntrinsic<[], - [ llvm_anyvector_ty, // Vector of pointers - llvm_anyint_ty, // Increment - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>], // Mask - [ IntrArgMemOnly ]>; + [ llvm_anyvector_ty, // Vector of pointers + llvm_anyint_ty, // Increment + LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>], // Mask + [ IntrArgMemOnly ]>; def int_experimental_vector_histogram_umin : DefaultAttrsIntrinsic<[], - [ llvm_anyvector_ty, // Vector of pointers - llvm_anyint_ty, // Update value - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>], // Mask - [ IntrArgMemOnly ]>; + [ llvm_anyvector_ty, // Vector of pointers + llvm_anyint_ty, // Update value + LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>], // Mask + [ IntrArgMemOnly ]>; def int_experimental_vector_histogram_umax : DefaultAttrsIntrinsic<[], - [ llvm_anyvector_ty, // Vector of pointers - llvm_anyint_ty, // Update value - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>], // Mask - [ IntrArgMemOnly ]>; + [ llvm_anyvector_ty, // Vector of pointers + llvm_anyint_ty, // Update value + LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>], // Mask + [ IntrArgMemOnly ]>; // Experimental match def int_experimental_vector_match : DefaultAttrsIntrinsic< - [ LLVMScalarOrSameVectorWidth<0, llvm_i1_ty> ], - [ llvm_anyvector_ty, - llvm_anyvector_ty, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty> ], // Mask - [ IntrNoMem, IntrSpeculatable ]>; + [ LLVMScalarOrSameVectorWidth<0, llvm_i1_ty> ], + [ llvm_anyvector_ty, + llvm_anyvector_ty, + LLVMScalarOrSameVectorWidth<0, llvm_i1_ty> ], // Mask + [ IntrNoMem, IntrSpeculatable ]>; // Extract based on mask bits def int_experimental_vector_extract_last_active: @@ -2112,305 +2270,12 @@ def int_experimental_vector_extract_last_active: // Operators let IntrProperties = [IntrNoMem, IntrSpeculatable] in { - // Integer arithmetic - def int_vp_add : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_sub : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_mul : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_ashr : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_lshr : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_shl : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_or : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_and : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_xor : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_abs : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - llvm_i1_ty, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_smin : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_smax : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_umin : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_umax : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_bswap : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_bitreverse : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_ctpop : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_fshl : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMMatchType<0>, - LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_fshr : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMMatchType<0>, - LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_sadd_sat : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_uadd_sat : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_ssub_sat : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_usub_sat : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - - // Floating-point arithmetic - def int_vp_fadd : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_fsub : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_fmul : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_fdiv : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_frem : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_fneg : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_fabs : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_sqrt : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_fma : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMMatchType<0>, - LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_fmuladd : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMMatchType<0>, - LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_minnum : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_maxnum : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_minimum : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_maximum : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_copysign : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_ceil : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_floor : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_round : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_roundeven : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_roundtozero : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_rint : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_nearbyint : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_lrint : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ llvm_anyvector_ty, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_llrint : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ llvm_anyvector_ty, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - - // Casts - def int_vp_trunc : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ llvm_anyvector_ty, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_zext : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ llvm_anyvector_ty, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_sext : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ llvm_anyvector_ty, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_fptrunc : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ llvm_anyvector_ty, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_fpext : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ llvm_anyvector_ty, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_fptoui : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ llvm_anyvector_ty, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_fptosi : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ llvm_anyvector_ty, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_uitofp : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ llvm_anyvector_ty, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_sitofp : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ llvm_anyvector_ty, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_ptrtoint : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ llvm_anyvector_ty, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_inttoptr : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ llvm_anyvector_ty, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - // Shuffles - def int_vp_select : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - LLVMMatchType<0>, - LLVMMatchType<0>, - llvm_i32_ty]>; def int_vp_merge : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], [ LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, LLVMMatchType<0>, LLVMMatchType<0>, llvm_i32_ty]>; - // Comparisons - def int_vp_fcmp : DefaultAttrsIntrinsic<[ LLVMScalarOrSameVectorWidth<0, llvm_i1_ty> ], - [ llvm_anyvector_ty, - LLVMMatchType<0>, - llvm_metadata_ty, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_icmp : DefaultAttrsIntrinsic<[ LLVMScalarOrSameVectorWidth<0, llvm_i1_ty> ], - [ llvm_anyvector_ty, - LLVMMatchType<0>, - llvm_metadata_ty, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - // Reductions def int_vp_reduce_fadd : DefaultAttrsIntrinsic<[LLVMVectorElementType<0>], [ LLVMVectorElementType<0>, @@ -2511,24 +2376,12 @@ def int_vp_urem : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, llvm_i32_ty], [IntrNoMem]>; -let IntrProperties = [IntrNoMem, IntrSpeculatable, ImmArg>] in { - def int_vp_ctlz : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - llvm_i1_ty, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - def int_vp_cttz : DefaultAttrsIntrinsic<[ llvm_anyvector_ty ], - [ LLVMMatchType<0>, - llvm_i1_ty, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty]>; - +let IntrProperties = [IntrNoMem, IntrSpeculatable, ImmArg>] in def int_vp_cttz_elts : DefaultAttrsIntrinsic<[ llvm_anyint_ty ], [ llvm_anyvector_ty, llvm_i1_ty, LLVMScalarOrSameVectorWidth<1, llvm_i1_ty>, llvm_i32_ty]>; -} def int_loop_dependence_raw_mask: DefaultAttrsIntrinsic<[llvm_anyvector_ty], @@ -2541,8 +2394,8 @@ def int_loop_dependence_war_mask: [IntrNoMem, IntrNoSync, IntrWillReturn, ImmArg>]>; def int_get_active_lane_mask: - DefaultAttrsIntrinsic<[llvm_anyvector_ty], - [llvm_anyint_ty, LLVMMatchType<1>], + DefaultAttrsIntrinsic<[llvm_any_vector_int_ty], + [llvm_any_scalar_int_ty, LLVMMatchType<1>], [IntrNoMem, IntrSpeculatable]>; def int_experimental_get_vector_length: @@ -2572,14 +2425,6 @@ def int_experimental_vp_reverse: llvm_i32_ty], [IntrNoMem, IntrSpeculatable]>; -def int_vp_is_fpclass: - DefaultAttrsIntrinsic<[ LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>], - [ llvm_anyvector_ty, - llvm_i32_ty, - LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, - llvm_i32_ty], - [IntrNoMem, IntrSpeculatable, ImmArg>]>; - //===-------------------------- Masked Intrinsics -------------------------===// // def int_masked_load: @@ -2608,75 +2453,83 @@ def int_masked_scatter: def int_masked_expandload: DefaultAttrsIntrinsic<[llvm_anyvector_ty], - [llvm_ptr_ty, LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, + [llvm_anyptr_ty, LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, LLVMMatchType<0>], [IntrReadMem, NoCapture>]>; def int_masked_compressstore: DefaultAttrsIntrinsic<[], - [llvm_anyvector_ty, llvm_ptr_ty, + [llvm_anyvector_ty, llvm_anyptr_ty, LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>], [IntrWriteMem, IntrArgMemOnly, NoCapture>]>; def int_experimental_vector_compress: DefaultAttrsIntrinsic<[llvm_anyvector_ty], - [LLVMMatchType<0>, LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, LLVMMatchType<0>], + [LLVMMatchType<0>, LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>, + LLVMMatchType<0>], [IntrNoMem]>; def int_masked_udiv: - DefaultAttrsIntrinsic<[llvm_anyvector_ty], + DefaultAttrsIntrinsic<[llvm_any_vector_int_ty], [LLVMMatchType<0>, LLVMMatchType<0>, LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>], [IntrNoMem]>; def int_masked_sdiv: - DefaultAttrsIntrinsic<[llvm_anyvector_ty], + DefaultAttrsIntrinsic<[llvm_any_vector_int_ty], [LLVMMatchType<0>, LLVMMatchType<0>, LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>], [IntrNoMem]>; def int_masked_urem: - DefaultAttrsIntrinsic<[llvm_anyvector_ty], + DefaultAttrsIntrinsic<[llvm_any_vector_int_ty], [LLVMMatchType<0>, LLVMMatchType<0>, LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>], [IntrNoMem]>; def int_masked_srem: - DefaultAttrsIntrinsic<[llvm_anyvector_ty], + DefaultAttrsIntrinsic<[llvm_any_vector_int_ty], [LLVMMatchType<0>, LLVMMatchType<0>, LLVMScalarOrSameVectorWidth<0, llvm_i1_ty>], [IntrNoMem]>; // Test whether a pointer is associated with a type metadata identifier. -def int_type_test : DefaultAttrsIntrinsic<[llvm_i1_ty], [llvm_ptr_ty, llvm_metadata_ty], - [IntrNoMem, IntrSpeculatable]>; +def int_type_test : DefaultAttrsIntrinsic<[llvm_i1_ty], + [llvm_ptr_ty, llvm_metadata_ty], + [IntrNoMem, IntrSpeculatable]>; -// Safely loads a function pointer from a virtual table pointer using type metadata. +// Safely loads a function pointer from a virtual table pointer using type +// metadata. def int_type_checked_load : DefaultAttrsIntrinsic<[llvm_ptr_ty, llvm_i1_ty], - [llvm_ptr_ty, llvm_i32_ty, llvm_metadata_ty], - [IntrNoMem]>; - -// Safely loads a relative function pointer from a virtual table pointer using type metadata. -def int_type_checked_load_relative : DefaultAttrsIntrinsic<[llvm_ptr_ty, llvm_i1_ty], - [llvm_ptr_ty, llvm_i32_ty, llvm_metadata_ty], - [IntrNoMem]>; + [llvm_ptr_ty, llvm_i32_ty, llvm_metadata_ty], + [IntrNoMem]>; + +// Safely loads a relative function pointer from a virtual table pointer using +// type metadata. +def int_type_checked_load_relative : DefaultAttrsIntrinsic< + [llvm_ptr_ty, llvm_i1_ty], + [llvm_ptr_ty, llvm_i32_ty, + llvm_metadata_ty], + [IntrNoMem]>; // Test whether a pointer is associated with a type metadata identifier. Used // for public visibility classes that may later be refined to private // visibility. -def int_public_type_test : DefaultAttrsIntrinsic<[llvm_i1_ty], [llvm_ptr_ty, llvm_metadata_ty], +def int_public_type_test : DefaultAttrsIntrinsic<[llvm_i1_ty], + [llvm_ptr_ty, llvm_metadata_ty], [IntrNoMem, IntrSpeculatable]>; // Create a branch funnel that implements an indirect call to a limited set of // callees. This needs to be a musttail call. def int_icall_branch_funnel : DefaultAttrsIntrinsic<[], [llvm_vararg_ty], []>; -def int_load_relative: DefaultAttrsIntrinsic<[llvm_ptr_ty], [llvm_ptr_ty, llvm_anyint_ty], +def int_load_relative: DefaultAttrsIntrinsic<[llvm_ptr_ty], + [llvm_ptr_ty, llvm_anyint_ty], [IntrReadMem, IntrArgMemOnly]>; def int_asan_check_memaccess : @@ -2684,7 +2537,8 @@ def int_asan_check_memaccess : // Spin in an infinite loop (using instructions specified by the target) iff the // argument is true. Used to implement efficient conditional traps. -def int_cond_loop : Intrinsic<[], [llvm_i1_ty], [IntrNoMem, IntrHasSideEffects]>; +def int_cond_loop : Intrinsic<[], [llvm_i1_ty], + [IntrNoMem, IntrHasSideEffects]>; // HWASan intrinsics to test whether a pointer is addressable. //===----------------------------------------------------------------------===// @@ -2733,68 +2587,73 @@ def int_xray_typedevent : Intrinsic<[], [llvm_i64_ty, llvm_ptr_ty, llvm_i64_ty], // @llvm.memcpy.element.unordered.atomic.*(dest, src, length, elementsize) def int_memcpy_element_unordered_atomic - : Intrinsic<[], - [llvm_anyptr_ty, llvm_anyptr_ty, llvm_anyint_ty, llvm_i32_ty], - [IntrArgMemOnly, IntrWillReturn, IntrNoSync, - NoCapture>, NoCapture>, - WriteOnly>, ReadOnly>, - ImmArg>]>; + : DefaultAttrsIntrinsic< + [], [llvm_anyptr_ty, llvm_anyptr_ty, llvm_anyint_ty, llvm_i32_ty], + [IntrArgMemOnly, NoCapture>, NoCapture>, + WriteOnly>, ReadOnly>, ImmArg>]>; // @llvm.memmove.element.unordered.atomic.*(dest, src, length, elementsize) def int_memmove_element_unordered_atomic - : Intrinsic<[], - [llvm_anyptr_ty, llvm_anyptr_ty, llvm_anyint_ty, llvm_i32_ty], - [IntrArgMemOnly, IntrWillReturn, IntrNoSync, - NoCapture>, NoCapture>, - WriteOnly>, ReadOnly>, - ImmArg>]>; + : DefaultAttrsIntrinsic< + [], [llvm_anyptr_ty, llvm_anyptr_ty, llvm_anyint_ty, llvm_i32_ty], + [IntrArgMemOnly, NoCapture>, NoCapture>, + WriteOnly>, ReadOnly>, ImmArg>]>; // @llvm.memset.element.unordered.atomic.*(dest, value, length, elementsize) def int_memset_element_unordered_atomic - : Intrinsic<[], [llvm_anyptr_ty, llvm_i8_ty, llvm_anyint_ty, llvm_i32_ty], - [IntrWriteMem, IntrArgMemOnly, IntrWillReturn, IntrNoSync, - NoCapture>, WriteOnly>, - ImmArg>]>; + : DefaultAttrsIntrinsic< + [], [llvm_anyptr_ty, llvm_i8_ty, llvm_anyint_ty, llvm_i32_ty], + [IntrWriteMem, IntrArgMemOnly, NoCapture>, + WriteOnly>, ImmArg>]>; //===------------------------ Reduction Intrinsics ------------------------===// // -let IntrProperties = [IntrNoMem, IntrSpeculatable, IntrNoCreateUndefOrPoison] in { +let IntrProperties = [IntrNoMem, IntrSpeculatable, IntrNoCreateUndefOrPoison] + in { def int_vector_reduce_fadd : DefaultAttrsIntrinsic<[LLVMVectorElementType<0>], [LLVMVectorElementType<0>, - llvm_anyvector_ty]>; + llvm_any_vector_float_ty]>; def int_vector_reduce_fmul : DefaultAttrsIntrinsic<[LLVMVectorElementType<0>], [LLVMVectorElementType<0>, - llvm_anyvector_ty]>; + llvm_any_vector_float_ty]>; def int_vector_reduce_add : DefaultAttrsIntrinsic<[LLVMVectorElementType<0>], - [llvm_anyvector_ty]>; + [llvm_any_vector_int_ty]>; def int_vector_reduce_mul : DefaultAttrsIntrinsic<[LLVMVectorElementType<0>], - [llvm_anyvector_ty]>; + [llvm_any_vector_int_ty]>; def int_vector_reduce_and : DefaultAttrsIntrinsic<[LLVMVectorElementType<0>], - [llvm_anyvector_ty]>; + [llvm_any_vector_int_ty]>; def int_vector_reduce_or : DefaultAttrsIntrinsic<[LLVMVectorElementType<0>], - [llvm_anyvector_ty]>; + [llvm_any_vector_int_ty]>; def int_vector_reduce_xor : DefaultAttrsIntrinsic<[LLVMVectorElementType<0>], - [llvm_anyvector_ty]>; + [llvm_any_vector_int_ty]>; def int_vector_reduce_smax : DefaultAttrsIntrinsic<[LLVMVectorElementType<0>], - [llvm_anyvector_ty]>; + [llvm_any_vector_int_ty]>; def int_vector_reduce_smin : DefaultAttrsIntrinsic<[LLVMVectorElementType<0>], - [llvm_anyvector_ty]>; + [llvm_any_vector_int_ty]>; def int_vector_reduce_umax : DefaultAttrsIntrinsic<[LLVMVectorElementType<0>], - [llvm_anyvector_ty]>; + [llvm_any_vector_int_ty]>; def int_vector_reduce_umin : DefaultAttrsIntrinsic<[LLVMVectorElementType<0>], - [llvm_anyvector_ty]>; + [llvm_any_vector_int_ty]>; def int_vector_reduce_fmax : DefaultAttrsIntrinsic<[LLVMVectorElementType<0>], - [llvm_anyvector_ty]>; + [llvm_any_vector_float_ty]>; def int_vector_reduce_fmin : DefaultAttrsIntrinsic<[LLVMVectorElementType<0>], - [llvm_anyvector_ty]>; - def int_vector_reduce_fminimum: DefaultAttrsIntrinsic<[LLVMVectorElementType<0>], - [llvm_anyvector_ty]>; - def int_vector_reduce_fmaximum: DefaultAttrsIntrinsic<[LLVMVectorElementType<0>], - [llvm_anyvector_ty]>; + [llvm_any_vector_float_ty]>; + def int_vector_reduce_fminimum: DefaultAttrsIntrinsic< + [LLVMVectorElementType<0>], + [llvm_any_vector_float_ty]>; + def int_vector_reduce_fmaximum: DefaultAttrsIntrinsic< + [LLVMVectorElementType<0>], + [llvm_any_vector_float_ty]>; + def int_vector_reduce_fminimumnum: DefaultAttrsIntrinsic< + [LLVMVectorElementType<0>], + [llvm_any_vector_float_ty]>; + def int_vector_reduce_fmaximumnum: DefaultAttrsIntrinsic< + [LLVMVectorElementType<0>], + [llvm_any_vector_float_ty]>; } -//===----- Matrix intrinsics ---------------------------------------------===// +//===------ Matrix intrinsics ---------------------------------------------===// def int_matrix_transpose : DefaultAttrsIntrinsic<[llvm_anyvector_ty], @@ -2834,8 +2693,10 @@ def int_set_loop_iterations : // Same as the above, but produces a value (the same as the input operand) to // be fed into the loop. -def int_start_loop_iterations : - DefaultAttrsIntrinsic<[llvm_anyint_ty], [LLVMMatchType<0>], [IntrNoDuplicate]>; +def int_start_loop_iterations : DefaultAttrsIntrinsic< + [llvm_anyint_ty], + [LLVMMatchType<0>], + [IntrNoDuplicate]>; // Specify that the value given is the number of iterations that the next loop // will execute. Also test that the given count is not zero, allowing it to @@ -2889,9 +2750,9 @@ def int_preserve_struct_access_index : DefaultAttrsIntrinsic<[llvm_anyptr_ty], ImmArg>, ImmArg>]>; def int_preserve_static_offset : DefaultAttrsIntrinsic<[llvm_ptr_ty], - [llvm_ptr_ty], - [IntrNoMem, IntrSpeculatable, - ReadNone >]>; + [llvm_ptr_ty], + [IntrNoMem, IntrSpeculatable, + ReadNone >]>; //===------------ Intrinsics to perform common vector shuffles ------------===// @@ -2918,23 +2779,22 @@ def int_vscale : DefaultAttrsIntrinsic<[llvm_anyint_ty], //===---------- Intrinsics to perform subvector insertion/extraction ------===// def int_vector_insert : DefaultAttrsIntrinsic<[llvm_anyvector_ty], - [LLVMMatchType<0>, llvm_anyvector_ty, llvm_i64_ty], - [IntrNoMem, IntrSpeculatable, ImmArg>]>; + [LLVMMatchType<0>, llvm_anyvector_ty, llvm_i64_ty], + [IntrNoMem, IntrSpeculatable, ImmArg>]>; def int_vector_extract : DefaultAttrsIntrinsic<[llvm_anyvector_ty], - [llvm_anyvector_ty, llvm_i64_ty], - [IntrNoMem, IntrSpeculatable, ImmArg>]>; + [llvm_anyvector_ty, llvm_i64_ty], + [IntrNoMem, IntrSpeculatable, ImmArg>]>; foreach n = 2...8 in { - def int_vector_interleave#n : DefaultAttrsIntrinsic<[llvm_anyvector_ty], - !listsplat(LLVMOneNthElementsVectorType<0, n>, n), - [IntrNoMem, - IntrSpeculatable]>; - - def int_vector_deinterleave#n : DefaultAttrsIntrinsic, n), - [llvm_anyvector_ty], - [IntrNoMem, - IntrSpeculatable]>; + def int_vector_interleave#n : DefaultAttrsIntrinsic<[llvm_anyvector_ty], + !listsplat(LLVMOneNthElementsVectorType<0, n>, n), + [IntrNoMem, IntrSpeculatable]>; + + def int_vector_deinterleave#n : DefaultAttrsIntrinsic< + !listsplat(LLVMOneNthElementsVectorType<0, n>, n), + [llvm_anyvector_ty], + [IntrNoMem, IntrSpeculatable]>; } //===-------------- Intrinsics to perform partial reduction ---------------===// @@ -2946,8 +2806,8 @@ def int_vector_partial_reduce_add : DefaultAttrsIntrinsic<[LLVMMatchType<0>], IntrSpeculatable]>; def int_vector_partial_reduce_fadd : DefaultAttrsIntrinsic<[LLVMMatchType<0>], - [llvm_anyfloat_ty, llvm_anyfloat_ty], - [IntrNoMem]>; + [llvm_anyfloat_ty, llvm_anyfloat_ty], + [IntrNoMem]>; //===----------------- Pointer Authentication Intrinsics ------------------===// // @@ -2979,6 +2839,18 @@ def int_ptrauth_resign : Intrinsic<[llvm_i64_ty], [IntrNoMem, ImmArg>, ImmArg>]>; +// Authenticate a signed pointer using a PC-based signature and resign it. +// The second (key) and third (discriminator) arguments specify the signing +// schema used for authenticating. +// The fourth argument specifies the PC value used for authenticating. +// The fifth and sixth arguments specify the schema used for resigning. +// The signature must be valid. +def int_ptrauth_auth_with_pc_and_resign : Intrinsic<[llvm_i64_ty], + [llvm_i64_ty, llvm_i32_ty, llvm_i64_ty, + llvm_i64_ty, llvm_i32_ty, llvm_i64_ty], + [IntrNoMem, ImmArg>, + ImmArg>]>; + // Authenticate a signed pointer, load 32bit value at offset from pointer, add // both, and sign it. The second (key) and third (discriminator) arguments // specify the signing schema used for authenticating. The fourth and fifth @@ -3051,6 +2923,7 @@ include "llvm/IR/IntrinsicsXCore.td" include "llvm/IR/IntrinsicsHexagon.td" include "llvm/IR/IntrinsicsNVVM.td" include "llvm/IR/IntrinsicsMips.td" +include "llvm/IR/IntrinsicsAVR.td" include "llvm/IR/IntrinsicsAMDGPU.td" include "llvm/IR/IntrinsicsBPF.td" include "llvm/IR/IntrinsicsSystemZ.td" diff --git a/examples/llvm/IR/IntrinsicsX86.td b/examples/llvm/IR/IntrinsicsX86.td index b75a048..ad79a8d 100644 --- a/examples/llvm/IR/IntrinsicsX86.td +++ b/examples/llvm/IR/IntrinsicsX86.td @@ -2575,18 +2575,6 @@ let TargetPrefix = "x86" in { // All intrinsics start with "llvm.x86.". def int_x86_bmi_bzhi_64 : ClangBuiltin<"__builtin_ia32_bzhi_di">, DefaultAttrsIntrinsic<[llvm_i64_ty], [llvm_i64_ty, llvm_i64_ty], [IntrNoMem]>; - def int_x86_bmi_pdep_32 : ClangBuiltin<"__builtin_ia32_pdep_si">, - DefaultAttrsIntrinsic<[llvm_i32_ty], [llvm_i32_ty, llvm_i32_ty], - [IntrNoMem]>; - def int_x86_bmi_pdep_64 : ClangBuiltin<"__builtin_ia32_pdep_di">, - DefaultAttrsIntrinsic<[llvm_i64_ty], [llvm_i64_ty, llvm_i64_ty], - [IntrNoMem]>; - def int_x86_bmi_pext_32 : ClangBuiltin<"__builtin_ia32_pext_si">, - DefaultAttrsIntrinsic<[llvm_i32_ty], [llvm_i32_ty, llvm_i32_ty], - [IntrNoMem]>; - def int_x86_bmi_pext_64 : ClangBuiltin<"__builtin_ia32_pext_di">, - DefaultAttrsIntrinsic<[llvm_i64_ty], [llvm_i64_ty, llvm_i64_ty], - [IntrNoMem]>; } //===----------------------------------------------------------------------===// @@ -5664,16 +5652,6 @@ let TargetPrefix = "x86" in { [llvm_i16_ty, llvm_i16_ty, llvm_x86amx_ty, llvm_i32_ty], []>; - def int_x86_tmmultf32ps : ClangBuiltin<"__builtin_ia32_tmmultf32ps">, - Intrinsic<[], [llvm_i8_ty, llvm_i8_ty, llvm_i8_ty], - [ImmArg>, ImmArg>, - ImmArg>]>; - def int_x86_tmmultf32ps_internal : - ClangBuiltin<"__builtin_ia32_tmmultf32ps_internal">, - Intrinsic<[llvm_x86amx_ty], - [llvm_i16_ty, llvm_i16_ty, llvm_i16_ty, llvm_x86amx_ty, - llvm_x86amx_ty, llvm_x86amx_ty], []>; - def int_x86_tdpbf8ps_internal : ClangBuiltin<"__builtin_ia32_tdpbf8ps_internal">, Intrinsic<[llvm_x86amx_ty], @@ -7341,4 +7319,26 @@ def int_x86_movrsdi : ClangBuiltin<"__builtin_ia32_movrsdi">, [IntrReadMem]>; def int_x86_prefetchrs : ClangBuiltin<"__builtin_ia32_prefetchrs">, Intrinsic<[], [llvm_ptr_ty], []>; + +//===----------------------------------------------------------------------===// +// BMM intrinsics + +def int_x86_avx512_vbmacor_v16hi : + ClangBuiltin<"__builtin_ia32_bmacor16x16x16_v16hi">, + DefaultAttrsIntrinsic<[llvm_v16i16_ty], [llvm_v16i16_ty, llvm_v16i16_ty, llvm_v16i16_ty], + [IntrNoMem]>; +def int_x86_avx512_vbmacor_v32hi : + ClangBuiltin<"__builtin_ia32_bmacor16x16x16_v32hi">, + DefaultAttrsIntrinsic<[llvm_v32i16_ty], [llvm_v32i16_ty, llvm_v32i16_ty, llvm_v32i16_ty], + [IntrNoMem]>; + +def int_x86_avx512_vbmacxor_v16hi : + ClangBuiltin<"__builtin_ia32_bmacxor16x16x16_v16hi">, + DefaultAttrsIntrinsic<[llvm_v16i16_ty], [llvm_v16i16_ty, llvm_v16i16_ty, llvm_v16i16_ty], + [IntrNoMem]>; +def int_x86_avx512_vbmacxor_v32hi : + ClangBuiltin<"__builtin_ia32_bmacxor16x16x16_v32hi">, + DefaultAttrsIntrinsic<[llvm_v32i16_ty], [llvm_v32i16_ty, llvm_v32i16_ty, llvm_v32i16_ty], + [IntrNoMem]>; } +//===----------------------------------------------------------------------===// diff --git a/examples/llvm/Target/AArch64/AArch64.td b/examples/llvm/Target/AArch64/AArch64.td index 20dea4b..9750fa8 100644 --- a/examples/llvm/Target/AArch64/AArch64.td +++ b/examples/llvm/Target/AArch64/AArch64.td @@ -50,6 +50,13 @@ def AArch64InstrInfo : InstrInfo; include "AArch64SystemOperands.td" +//===----------------------------------------------------------------------===// +// LFI rewriter metadata consumed by AArch64MCLFIRewriter. Must follow +// AArch64SystemOperands.td due to use of SearchableTable.td. +//===----------------------------------------------------------------------===// + +include "AArch64LFI.td" + //===----------------------------------------------------------------------===// // AArch64 Processors supported. // @@ -60,14 +67,14 @@ include "AArch64SystemOperands.td" class AArch64Unsupported { list F; } -let F = [HasSVE2p1, HasSVE2p1_or_SME2, HasSVE2p1_or_StreamingSME2, HasSVE2p1_or_SME2p1] in +let F = [HasSVE2p1, HasSVE2p1_or_SME, HasSVE2p1_or_SME2, HasSVE2p1_or_StreamingSME2, HasSVE2p1_or_SME2p1] in def SVE2p1Unsupported : AArch64Unsupported; def SVE2Unsupported : AArch64Unsupported { let F = !listconcat([HasSVE2, HasSVE2_or_SME, HasNonStreamingSVE2_or_SME2, HasSSVE_FP8FMA, HasSMEF8F16, HasSSVE_FP8DOT2, HasSSVE_FP8DOT4, HasSMEF8F32, HasSVEAES, HasSVESHA3, HasSVESM4, HasSVEBitPerm, - HasSVEB16B16], + HasSVEB16B16, HasSVEBFSCALE], SVE2p1Unsupported.F); } @@ -98,12 +105,13 @@ def SME2Unsupported : AArch64Unsupported { let F = !listconcat([HasSME2, HasNonStreamingSVE2_or_SME2, HasSVE2p1_or_SME2, HasSSVE_FP8FMA, HasSSVE_FP8DOT2, HasSSVE_FP8DOT4, HasSMEF8F16, HasSMEF8F32, HasSMEF16F16_or_SMEF8F16, HasSMEB16B16, - HasNonStreamingSVE_or_SSVE_AES, HasSVE2p1_or_StreamingSME2], + HasNonStreamingSVE_or_SSVE_AES, HasSVE2p1_or_StreamingSME2, HasSME2andIsNonStreamingSafe], SME2p1Unsupported.F); } def SMEUnsupported : AArch64Unsupported { - let F = !listconcat([HasSME, HasSMEI16I64, HasSMEF16F16, HasSMEF64F64, HasSMEFA64], + let F = !listconcat([HasSME, HasSMEI16I64, HasSMEF16F16, HasSMEF64F64, HasSMEFA64, + HasSMEandIsNonStreamingSafe], SME2Unsupported.F); } @@ -119,6 +127,7 @@ include "AArch64SchedA53.td" include "AArch64SchedA55.td" include "AArch64SchedA510.td" include "AArch64SchedA57.td" +include "AArch64SchedC1Nano.td" include "AArch64SchedC1Ultra.td" include "AArch64SchedC1Premium.td" include "AArch64SchedCyclone.td" @@ -132,6 +141,7 @@ include "AArch64SchedThunderX2T99.td" include "AArch64SchedA64FX.td" include "AArch64SchedThunderX3T110.td" include "AArch64SchedTSV110.td" +include "AArch64SchedHIP12.td" include "AArch64SchedAmpere1.td" include "AArch64SchedAmpere1B.td" include "AArch64SchedNeoverseN1.td" diff --git a/examples/llvm/Target/AArch64/AArch64InstrInfo.td b/examples/llvm/Target/AArch64/AArch64InstrInfo.td index fb54757..f09f9c2 100644 --- a/examples/llvm/Target/AArch64/AArch64InstrInfo.td +++ b/examples/llvm/Target/AArch64/AArch64InstrInfo.td @@ -219,6 +219,8 @@ def HasLSUI : Predicate<"Subtarget->hasLSUI()">, AssemblerPredicateWithAll<(all_of FeatureLSUI), "lsui">; def HasOCCMO : Predicate<"Subtarget->hasOCCMO()">, AssemblerPredicateWithAll<(all_of FeatureOCCMO), "occmo">; +def HasHINTE : Predicate<"Subtarget->hasHINTE()">, + AssemblerPredicateWithAll<(all_of FeatureHINTE), "hinte">; def HasLSCP : Predicate<"Subtarget->hasLSCP()">, AssemblerPredicateWithAll<(all_of FeatureLSCP), "lscp">; def HasSVE2p2 : Predicate<"Subtarget->isSVEAvailable() && Subtarget->hasSVE2p2()">, @@ -227,9 +229,9 @@ def HasSVE_B16MM : Predicate<"Subtarget->isSVEAvailable() && Subtarget->hasS AssemblerPredicateWithAll<(all_of FeatureSVE_B16MM), "sve-b16mm">; def HasF16MM : Predicate<"Subtarget->hasF16MM()">, AssemblerPredicateWithAll<(all_of FeatureF16MM), "f16mm">; -def HasSVE2p3 : Predicate<"Subtarget->hasSVE2p3()">, +def HasSVE2p3 : Predicate<"Subtarget->isSVEAvailable() && Subtarget->hasSVE2p3()">, AssemblerPredicateWithAll<(all_of FeatureSVE2p3), "sve2p3">; -def HasSME2p3 : Predicate<"Subtarget->hasSME2p3()">, +def HasSME2p3 : Predicate<"Subtarget->isStreaming() && Subtarget->hasSME2p3()">, AssemblerPredicateWithAll<(all_of FeatureSME2p3), "sme2p3">; def HasF16F32DOT : Predicate<"Subtarget->hasF16F32DOT()">, AssemblerPredicateWithAll<(all_of FeatureF16F32DOT), "f16f32dot">; @@ -434,6 +436,10 @@ def AArch64LocalRecover : SDNode<"ISD::LOCAL_RECOVER", def AllowMisalignedMemAccesses : Predicate<"!Subtarget->requiresStrictAlign()">; +def DisallowMisalignedMemAccesses + : Predicate<"Subtarget->requiresStrictAlign()">; +def IsBEOrDisallowMisalignedMemAccesses + : Predicate<"!Subtarget->isLittleEndian() || Subtarget->requiresStrictAlign()">; def UseWzrToVecMove : Predicate<"Subtarget->useWzrToVecMove()">; @@ -507,6 +513,9 @@ def SDT_AArch64Insr : SDTypeProfile<1, 2, [SDTCisVec<0>]>; def SDT_AArch64Zip : SDTypeProfile<1, 2, [SDTCisVec<0>, SDTCisSameAs<0, 1>, SDTCisSameAs<0, 2>]>; +def SDT_AArch64Addhn : SDTypeProfile<1, 2, [SDTCisVec<0>, SDTCisVec<1>, + SDTCisSameAs<1, 2>, + SDTCisOpSmallerThanOp<0, 1>]>; def SDT_AArch64MOVIedit : SDTypeProfile<1, 1, [SDTCisInt<1>]>; def SDT_AArch64MOVIshift : SDTypeProfile<1, 2, [SDTCisInt<1>, SDTCisInt<2>]>; def SDT_AArch64vecimm : SDTypeProfile<1, 3, [SDTCisVec<0>, SDTCisSameAs<0,1>, @@ -522,11 +531,7 @@ def SDT_AArch64vshiftinsert : SDTypeProfile<1, 3, [SDTCisVec<0>, SDTCisInt<3>, SDTCisSameAs<0,1>, SDTCisSameAs<0,2>]>; -def SDT_AArch64unvec : SDTypeProfile<1, 1, [SDTCisVec<0>, SDTCisSameAs<0,1>]>; -def SDT_AArch64fcmpz : SDTypeProfile<1, 1, []>; def SDT_AArch64fcmp : SDTypeProfile<1, 2, [SDTCisSameAs<1,2>]>; -def SDT_AArch64binvec : SDTypeProfile<1, 2, [SDTCisVec<0>, SDTCisSameAs<0,1>, - SDTCisSameAs<0,2>]>; def SDT_AArch64trivec : SDTypeProfile<1, 3, [SDTCisVec<0>, SDTCisSameAs<0,1>, SDTCisSameAs<0,2>, SDTCisSameAs<0,3>]>; @@ -536,9 +541,6 @@ def SDT_AArch64RANGE_PREFETCH: SDTypeProfile<0, 3, [SDTCisVT<0, i32>, SDTCisPtrT def SDT_AArch64ITOF : SDTypeProfile<1, 1, [SDTCisFP<0>, SDTCisSameAs<0,1>]>; -def SDT_AArch64TLSDescCall : SDTypeProfile<0, -2, [SDTCisPtrTy<0>, - SDTCisPtrTy<1>]>; - def SDT_AArch64uaddlp : SDTypeProfile<1, 1, [SDTCisVec<0>, SDTCisVec<1>]>; def SDT_AArch64ldp : SDTypeProfile<2, 1, [SDTCisVT<0, i64>, SDTCisSameAs<0, 1>, SDTCisPtrTy<2>]>; @@ -767,6 +769,35 @@ def topbitsallzero64: PatLeaf<(i64 GPR64:$src), [{ return VT && VT->maskedValueIsZero(Reg, APInt::getHighBitsSet(64, 63)); }]; } +// Loads and stores with a minimum alignment. +class load_aligned : PatFrag<(ops node:$ptr), (unindexedload node:$ptr)> { + let IsLoad = true; + let IsNonExtLoad = true; + let MinAlignment = align_bytes; +} +class store_aligned : PatFrag<(ops node:$val, node:$ptr), + (unindexedstore node:$val, node:$ptr)> { + let IsStore = true; + let IsTruncStore = false; + let MinAlignment = align_bytes; +} +class pre_store_aligned : PatFrag<(ops node:$val, node:$base, node:$offset), + (istore node:$val, node:$base, node:$offset), [{ + ISD::MemIndexedMode AM = cast(N)->getAddressingMode(); + Align A = cast(N)->getAlign(); + return (AM == ISD::PRE_INC || AM == ISD::PRE_DEC) && A >= }] # align_bytes # [{; +}]> { + let MinAlignment = align_bytes; +} +class post_store_aligned : PatFrag<(ops node:$val, node:$ptr, node:$offset), + (istore node:$val, node:$ptr, node:$offset), [{ + ISD::MemIndexedMode AM = cast(N)->getAddressingMode(); + Align A = cast(N)->getAlign(); + return (AM == ISD::POST_INC || AM == ISD::POST_DEC) && A >= }] # align_bytes # [{; +}]> { + let MinAlignment = align_bytes; +} + // Node definitions. // Compare-and-branch def AArch64CB : SDNode<"AArch64ISD::CB", SDT_AArch64cb, [SDNPHasChain]>; @@ -955,6 +986,9 @@ def AArch64vsri : SDNode<"AArch64ISD::VSRI", SDT_AArch64vshiftinsert>; // element must be identical. def AArch64bsp: SDNode<"AArch64ISD::BSP", SDT_AArch64trivec>; +// AArch64ISD::CMTST node: result is all-ones per lane where (X & Y) != 0. +def AArch64cmtst: SDNode<"AArch64ISD::CMTST", SDT_AArch64Zip>; + def AArch64cmeq : PatFrag<(ops node:$lhs, node:$rhs), (setcc node:$lhs, node:$rhs, SETEQ)>; def AArch64cmge : PatFrag<(ops node:$lhs, node:$rhs), @@ -983,9 +1017,6 @@ def AArch64cmlez : PatFrag<(ops node:$lhs), def AArch64cmltz : PatFrag<(ops node:$lhs), (setcc immAllZerosV, node:$lhs, SETGT)>; -def AArch64cmtst : PatFrag<(ops node:$LHS, node:$RHS), - (vnot (AArch64cmeqz (and node:$LHS, node:$RHS)))>; - def AArch64fcmeqz : PatFrag<(ops node:$lhs), (AArch64fcmeq node:$lhs, immAllZerosV)>; @@ -1113,6 +1144,8 @@ def AArch64usdot : SDNode<"AArch64ISD::USDOT", SDT_AArch64Dot>; def AArch64saddv : SDNode<"AArch64ISD::SADDV", SDT_AArch64UnaryVec>; def AArch64uaddv : SDNode<"AArch64ISD::UADDV", SDT_AArch64UnaryVec>; +def AArch64addhn : SDNode<"AArch64ISD::ADDHN", SDT_AArch64Addhn>; + // Vector across-lanes min/max // Only the lower result lane is defined. def AArch64sminv : SDNode<"AArch64ISD::SMINV", SDT_AArch64UnaryVec>; @@ -1409,6 +1442,15 @@ def PROBED_STACKALLOC_DYN : Pseudo<(outs), } // Defs = [SP, NZCV], Uses = [SP] in } // hasSideEffects = 1, isCodeGenOnly = 1 +// Read of an allocatable register by name (e.g. the MSVC __getReg/__getRegFp +// intrinsics, lowered from llvm.read[_volatile]_register). +let hasSideEffects = 1, mayLoad = 1, Size = 4, isCodeGenOnly = 1 in { +def READ_REGISTER_GPR64 : Pseudo<(outs GPR64:$Rt), (ins i32imm:$reg), []>, + Sched<[]>; +def READ_REGISTER_FPR64 : Pseudo<(outs FPR64:$Rt), (ins i32imm:$reg), []>, + Sched<[]>; +} + let isReMaterializable = 1, isCodeGenOnly = 1 in { // FIXME: The following pseudo instructions are only needed because remat // cannot handle multiple instructions. When that changes, they can be @@ -1550,13 +1592,16 @@ let hasSideEffects = 1, isCodeGenOnly = 1, isTerminator = 1, isBarrier = 1 in { //===----------------------------------------------------------------------===// def HINT : HintI<"hint">; -def : InstAlias<"yield",(HINT 0b001)>; -def : InstAlias<"wfe", (HINT 0b010)>; -def : InstAlias<"wfi", (HINT 0b011)>; -def : InstAlias<"sev", (HINT 0b100)>; -def : InstAlias<"sevl", (HINT 0b101)>; -def : InstAlias<"dgh", (HINT 0b110)>; -def : InstAlias<"esb", (HINT 0b10000)>, Requires<[HasRAS]>; +let Predicates = [HasHINTE] in +def HINTE : HintE<"hinte">; + +def : InstAlias<"yield",(HINT 1)>; +def : InstAlias<"wfe", (HINT 2)>; +def : InstAlias<"wfi", (HINT 3)>; +def : InstAlias<"sev", (HINT 4)>; +def : InstAlias<"sevl", (HINT 5)>; +def : InstAlias<"dgh", (HINT 6)>; +def : InstAlias<"esb", (HINT 16)>, Requires<[HasRAS]>; def : InstAlias<"csdb", (HINT 20)>; let CRm = 0b0000, hasSideEffects = 0 in @@ -1564,20 +1609,30 @@ def NOP : SystemNoOperands<0b000, "hint\t#0">; def : InstAlias<"nop", (NOP)>; -def STSHH: STSHHI; +def : InstAlias<"stshh keep", (HINT 48), 1>; +def : InstAlias<"stshh strm", (HINT 49), 1>; +def : InstAlias<"shuh", (HINT 50), 1>; +def : InstAlias<"shuh ph", (HINT 51), 1>; +def : InstAlias<"stcph", (HINT 52), 1>; // In order to be able to write readable assembly, LLVM should accept assembly // inputs that use Branch Target Identification mnemonics, even with BTI disabled. // However, in order to be compatible with other assemblers (e.g. GAS), LLVM // should not emit these mnemonics unless BTI is enabled. def : InstAlias<"bti", (HINT 32), 0>; -def : InstAlias<"bti $op", (HINT btihint_op:$op), 0>; +def : InstAlias<"bti r", (HINT 32), 0>; +def : InstAlias<"bti c", (HINT 34), 0>; +def : InstAlias<"bti j", (HINT 36), 0>; +def : InstAlias<"bti jc", (HINT 38), 0>; def : InstAlias<"bti r", (HINT 32)>, Requires<[HasBTIE]>; def : InstAlias<"bti", (HINT 32)>, Requires<[HasBTI]>; -def : InstAlias<"bti $op", (HINT btihint_op:$op)>, Requires<[HasBTI]>; +def : InstAlias<"bti c", (HINT 34)>, Requires<[HasBTI]>; +def : InstAlias<"bti j", (HINT 36)>, Requires<[HasBTI]>; +def : InstAlias<"bti jc", (HINT 38)>, Requires<[HasBTI]>; // v8.2a Statistical Profiling extension -def : InstAlias<"psb $op", (HINT psbhint_op:$op)>, Requires<[HasSPE]>; +def : InstAlias<"psb csync", (HINT 17), 1>, Requires<[HasSPE]>; +def : InstAlias<"tsb csync", (HINT 18), 1>, Requires<[HasTRACEV8_4]>; // As far as LLVM is concerned this writes to the system's exclusive monitors. let mayLoad = 1, mayStore = 1 in @@ -1595,12 +1650,6 @@ def DSB : CRmSystemI; -def TSB : CRmSystemI { - let CRm = 0b0010; - let Inst{12} = 0; - let Predicates = [HasTRACEV8_4]; -} - def DSBnXS : CRmSystemI { let CRm{1-0} = 0b11; let Inst{9-8} = 0b10; @@ -1612,16 +1661,6 @@ def WFET : RegInputSystemI<0b0000, 0b000, "wfet">; def WFIT : RegInputSystemI<0b0000, 0b001, "wfit">; } -// Branch Record Buffer two-word mnemonic instructions -class BRBEI op2, string keyword> - : SimpleSystemI<0, (ins), "brb", keyword>, Sched<[WriteSys]> { - let Inst{31-8} = 0b110101010000100101110010; - let Inst{7-5} = op2; - let Predicates = [HasBRBE]; -} -def BRB_IALL: BRBEI<0b100, "\tiall">; -def BRB_INJ: BRBEI<0b101, "\tinj">; - } // Allow uppercase and lowercase keyword arguments for BRB IALL and BRB INJ @@ -2266,6 +2305,8 @@ let Predicates = [HasPAuth] in { (ins i32imm:$AUTKey, i64imm:$AUTDisc, GPR64noip:$AUTAddrDisc, i32imm:$PACKey, i64imm:$PACDisc, GPR64noip:$PACAddrDisc), []>, Sched<[WriteI, ReadI]> { + // Thanks to its register class, $PACAddrDisc never aliases X16 which is + // early-clobbered. let isCodeGenOnly = 1; let hasSideEffects = 1; let mayStore = 0; @@ -2275,6 +2316,29 @@ let Predicates = [HasPAuth] in { let Uses = [X16]; } + // AUT a pointer using a PC-blended discriminator and re-PAC with different + // keys/data. This directly manipulates x16/x17, which are the only registers + // certain OSs guarantee are safe to use for sensitive operations. + // Additionally, it uses fixed implicit registers x15/x16/x17 for the AUT as + // AUTI[AB]171615 is the only instruction that can do the three-input-operand + // AUT, unlike the other PAuth-related pseudos (AUTx16x17, AUTPAC) where + // the register operands are structurally necessary. + def AUTPCPAC + : Pseudo<(outs), + (ins i32imm:$AUTKey, + i32imm:$PACKey, i64imm:$PACDisc, GPR64noip:$PACAddrDisc), + []>, Sched<[WriteI, ReadI]> { + // Thanks to its register class, $PACAddrDisc never aliases X17 which is + // early-clobbered. + let isCodeGenOnly = 1; + let hasSideEffects = 1; + let mayStore = 0; + let mayLoad = 0; + let Size = 48; + let Defs = [X17,X15,X16,NZCV]; + let Uses = [X15,X16,X17]; + } + // Similiar to AUTPAC, except a 32bit value is loaded at Addend offset from // pointer and this value is added to the pointer before signing. This // directly manipulates x16/x17, which are the only registers the OS @@ -2286,6 +2350,8 @@ let Predicates = [HasPAuth] in { i64imm:$Addend), []>, Sched<[WriteI, ReadI]> { + // Thanks to its register class, $PACAddrDisc never aliases X16 which is + // early-clobbered. let isCodeGenOnly = 1; let hasSideEffects = 1; let mayStore = 0; @@ -2537,10 +2603,9 @@ def MSR_FPSR : Pseudo<(outs), (ins GPR64:$val), PseudoInstExpansion<(MSR 0xda21, GPR64:$val)>, Sched<[WriteSys]>; -let Defs = [FPMR] in +let Uses = [FPMR], Defs = [FPMR, NZCV], usesCustomInserter = 1 in def MSR_FPMR : Pseudo<(outs), (ins GPR64:$val), [(int_aarch64_set_fpmr i64:$val)]>, - PseudoInstExpansion<(MSR 0xda22, GPR64:$val)>, Sched<[WriteSys]>; // Generic system instructions @@ -2550,6 +2615,14 @@ def SYSLxt : SystemLXtI<1, "sysl">; def : InstAlias<"sys $op1, $Cn, $Cm, $op2", (SYSxt timm32_0_7:$op1, sys_cr_op:$Cn, sys_cr_op:$Cm, timm32_0_7:$op2, XZR)>; +def : InstAlias<"apas $Xt", + (SYSxt 6, 7, 0, 0, GPR64:$Xt), 2>; +def : InstAlias<"brb\tiall", + (SYSxt 1, 7, 2, 4, XZR), 2>, Requires<[HasBRBE]>; +def : InstAlias<"brb\tinj", + (SYSxt 1, 7, 2, 5, XZR), 2>, Requires<[HasBRBE]>; +def : InstAlias<"trcit $Rt", + (SYSxt 3, 7, 2, 7, GPR64:$Rt), 2>, Requires<[HasITE]>; //===----------------------------------------------------------------------===// @@ -2759,6 +2832,12 @@ def copyFromSP: PatLeaf<(i64 GPR64:$src), [{ cast(N->getOperand(1))->getReg() == AArch64::SP; }]>; +// Pseudo instruction for stack guard cookie unmixing (epilogue). +// This will be expanded post-RA to use FP directly, avoiding an extra mov. +// Pattern matching for this will be done via custom C++ code during instruction selection. +def STACK_GUARD_UNMIX : Pseudo<(outs GPR64:$dst), (ins GPR64:$stored_val), []>, + Sched<[]>; + // Use SUBS instead of SUB to enable CSE between SUBS and SUB. def : Pat<(sub GPR32sp:$Rn, addsub_shifted_imm32:$imm), (SUBSWri GPR32sp:$Rn, addsub_shifted_imm32:$imm)>; @@ -2809,6 +2888,17 @@ def : Pat<(AArch64sub_flag GPR64:$Rn, neg_addsub_shifted_imm64:$imm), (ADDSXri GPR64:$Rn, neg_addsub_shifted_imm64:$imm)>; } +def negate_imm : SDNodeXFormgetTargetConstant(-N->getAPIntValue(), SDLoc(N), N->getValueType(0)); +}]>; +def neg_cheaper_imm32 : PatLeaf<(i32 imm), [{ return isWorthNegatingImm(Op); }], negate_imm>; +def neg_cheaper_imm64 : PatLeaf<(i64 imm), [{ return isWorthNegatingImm(Op); }], negate_imm>; + +// Prefer (sub x, -c) over (add x, c) if -c is cheaper to materialise than c. +def : Pat<(add GPR32:$Rn, neg_cheaper_imm32:$imm), + (SUBSWrr GPR32:$Rn, (MOVi32imm neg_cheaper_imm32:$imm))>; +def : Pat<(add GPR64:$Rn, neg_cheaper_imm64:$imm), + (SUBSXrr GPR64:$Rn, (MOVi64imm neg_cheaper_imm64:$imm))>; def trunc_isWorthFoldingALU : PatFrag<(ops node:$src), (trunc $src)> { let PredicateCode = [{ return isWorthFoldingALU(SDValue(N, 0)); }]; @@ -2895,9 +2985,9 @@ def : Pat<(i64 (mul (ineg GPR64:$Rn), GPR64:$Rm)), } // AddedComplexity = 5 let AddedComplexity = 5 in { -def SMADDLrrr : WideMulAccum<0, 0b001, "smaddl", add, sext>; +def SMADDLrrr : WideMulAccum<0, 0b001, "smaddl", add_like, sext>; def SMSUBLrrr : WideMulAccum<1, 0b001, "smsubl", sub, sext>; -def UMADDLrrr : WideMulAccum<0, 0b101, "umaddl", add, zext>; +def UMADDLrrr : WideMulAccum<0, 0b101, "umaddl", add_like, zext>; def UMSUBLrrr : WideMulAccum<1, 0b101, "umsubl", sub, zext>; def : Pat<(i64 (mul (sext_inreg GPR64:$Rn, i32), (sext_inreg GPR64:$Rm, i32))), @@ -2934,11 +3024,11 @@ def : Pat<(i64 (ineg (mul (sext_inreg GPR64:$Rn, i32), (s64imm_32bit:$C)))), (SMSUBLrrr (i32 (EXTRACT_SUBREG GPR64:$Rn, sub_32)), (MOVi32imm (trunc_imm imm:$C)), XZR)>; -def : Pat<(i64 (add (mul (sext GPR32:$Rn), (s64imm_32bit:$C)), GPR64:$Ra)), +def : Pat<(i64 (add_like (mul (sext GPR32:$Rn), (s64imm_32bit:$C)), GPR64:$Ra)), (SMADDLrrr GPR32:$Rn, (MOVi32imm (trunc_imm imm:$C)), GPR64:$Ra)>; -def : Pat<(i64 (add (mul (zext GPR32:$Rn), (i64imm_32bit:$C)), GPR64:$Ra)), +def : Pat<(i64 (add_like (mul (zext GPR32:$Rn), (i64imm_32bit:$C)), GPR64:$Ra)), (UMADDLrrr GPR32:$Rn, (MOVi32imm (trunc_imm imm:$C)), GPR64:$Ra)>; -def : Pat<(i64 (add (mul (sext_inreg GPR64:$Rn, i32), (s64imm_32bit:$C)), +def : Pat<(i64 (add_like (mul (sext_inreg GPR64:$Rn, i32), (s64imm_32bit:$C)), GPR64:$Ra)), (SMADDLrrr (i32 (EXTRACT_SUBREG GPR64:$Rn, sub_32)), (MOVi32imm (trunc_imm imm:$C)), GPR64:$Ra)>; @@ -2957,9 +3047,9 @@ def : Pat<(i64 (smullwithsignbits GPR64:$Rn, GPR64:$Rm)), def : Pat<(i64 (smullwithsignbits GPR64:$Rn, (sext GPR32:$Rm))), (SMADDLrrr (EXTRACT_SUBREG GPR64:$Rn, sub_32), $Rm, XZR)>; -def : Pat<(i64 (add (smullwithsignbits GPR64:$Rn, GPR64:$Rm), GPR64:$Ra)), +def : Pat<(i64 (add_like (smullwithsignbits GPR64:$Rn, GPR64:$Rm), GPR64:$Ra)), (SMADDLrrr (EXTRACT_SUBREG GPR64:$Rn, sub_32), (EXTRACT_SUBREG GPR64:$Rm, sub_32), GPR64:$Ra)>; -def : Pat<(i64 (add (smullwithsignbits GPR64:$Rn, (sext GPR32:$Rm)), GPR64:$Ra)), +def : Pat<(i64 (add_like (smullwithsignbits GPR64:$Rn, (sext GPR32:$Rm)), GPR64:$Ra)), (SMADDLrrr (EXTRACT_SUBREG GPR64:$Rn, sub_32), $Rm, GPR64:$Ra)>; def : Pat<(i64 (ineg (smullwithsignbits GPR64:$Rn, GPR64:$Rm))), @@ -2977,9 +3067,9 @@ def : Pat<(i64 (mul top32Zero:$Rn, top32Zero:$Rm)), def : Pat<(i64 (mul top32Zero:$Rn, (zext GPR32:$Rm))), (UMADDLrrr (EXTRACT_SUBREG GPR64:$Rn, sub_32), $Rm, XZR)>; -def : Pat<(i64 (add (mul top32Zero:$Rn, top32Zero:$Rm), GPR64:$Ra)), +def : Pat<(i64 (add_like (mul top32Zero:$Rn, top32Zero:$Rm), GPR64:$Ra)), (UMADDLrrr (EXTRACT_SUBREG GPR64:$Rn, sub_32), (EXTRACT_SUBREG GPR64:$Rm, sub_32), GPR64:$Ra)>; -def : Pat<(i64 (add (mul top32Zero:$Rn, (zext GPR32:$Rm)), GPR64:$Ra)), +def : Pat<(i64 (add_like (mul top32Zero:$Rn, (zext GPR32:$Rm)), GPR64:$Ra)), (UMADDLrrr (EXTRACT_SUBREG GPR64:$Rn, sub_32), $Rm, GPR64:$Ra)>; def : Pat<(i64 (ineg (mul top32Zero:$Rn, top32Zero:$Rm))), @@ -3079,9 +3169,6 @@ let Predicates = [HasLSUI] in { defm : STOPregisterLSUI<"sttset","LDTSET">; // STTSETx } -// v9.6-a FEAT_RME_GPC3 -def APAS : APASI; - // v8.1 atomic LD(register). Performs load and then ST(register) defm LDADD : LDOPregister<0b000, "add", 0, 0, "">; defm LDADDA : LDOPregister<0b000, "add", 1, 0, "a">; @@ -3748,6 +3835,17 @@ def : Pat<(AArch64call texternalsym:$func), (BL texternalsym:$func)>; //===----------------------------------------------------------------------===// // Exception generation instructions. //===----------------------------------------------------------------------===// + +// Nodes for the HVC/SVC exception-generating calls. The single operand is the +// 16-bit instruction immediate; the argument registers, clobber mask and glue +// are attached as variadic operands during lowering. +def AArch64hvc : SDNode<"AArch64ISD::HVC", SDTypeProfile<0, 1, [SDTCisInt<0>]>, + [SDNPHasChain, SDNPOptInGlue, SDNPOutGlue, SDNPVariadic, + SDNPMayLoad, SDNPMayStore]>; +def AArch64svc : SDNode<"AArch64ISD::SVC", SDTypeProfile<0, 1, [SDTCisInt<0>]>, + [SDNPHasChain, SDNPOptInGlue, SDNPOutGlue, SDNPVariadic, + SDNPMayLoad, SDNPMayStore]>; + let isTrap = 1 in { def BRK : ExceptionGeneration<0b001, 0b00, "brk", [(int_aarch64_break timm32_0_65535:$imm)]>; @@ -3757,9 +3855,18 @@ def DCPS2 : ExceptionGeneration<0b101, 0b10, "dcps2">; def DCPS3 : ExceptionGeneration<0b101, 0b11, "dcps3">, Requires<[HasEL3]>; def HLT : ExceptionGeneration<0b010, 0b00, "hlt", [(int_aarch64_hlt timm32_0_65535:$imm)]>; +// Following the Microsoft __hvc/__svc calling convention (a software +// convention, not an architectural rule), HVC and SVC pass up to four arguments +// in X0-X3 and return a value in X0. They are therefore selected from the +// AArch64hvc/AArch64svc nodes (built by LowerINTRINSIC_W_CHAIN) rather than a +// plain intrinsic pattern, and define X0 for the result. +let mayLoad = 1, mayStore = 1, Defs = [X0] in def HVC : ExceptionGeneration<0b000, 0b10, "hvc">; def SMC : ExceptionGeneration<0b000, 0b11, "smc">, Requires<[HasEL3]>; +let mayLoad = 1, mayStore = 1, Defs = [X0] in def SVC : ExceptionGeneration<0b000, 0b01, "svc">; +def : Pat<(AArch64hvc timm32_0_65535:$imm), (HVC timm32_0_65535:$imm)>; +def : Pat<(AArch64svc timm32_0_65535:$imm), (SVC timm32_0_65535:$imm)>; // DCPSn defaults to an immediate operand of zero if unspecified. def : InstAlias<"dcps1", (DCPS1 0)>; @@ -3866,43 +3973,48 @@ defm LDRSW : Load32RO<0b10, 0, 0b10, GPR64, "ldrsw", i64, sextloadi32>; // Pre-fetch. defm PRFM : PrefetchRO<0b11, 0, 0b10, "prfm">; -// Match all load 64 bits width whose type is compatible with FPR64 multiclass VecROLoadPat { - - def : Pat<(VecTy (load (ro.Wpat GPR64sp:$Rn, GPR32:$Rm, ro.Wext:$extend))), - (LOADW GPR64sp:$Rn, GPR32:$Rm, ro.Wext:$extend)>; + Instruction LOADW, Instruction LOADX, + int vector_align, list preds> { + let Predicates = !listconcat([AllowMisalignedMemAccesses], preds) in { + def : Pat<(VecTy (load (ro.Wpat GPR64sp:$Rn, GPR32:$Rm, ro.Wext:$extend))), + (LOADW GPR64sp:$Rn, GPR32:$Rm, ro.Wext:$extend)>; + + def : Pat<(VecTy (load (ro.Xpat GPR64sp:$Rn, GPR64:$Rm, ro.Xext:$extend))), + (LOADX GPR64sp:$Rn, GPR64:$Rm, ro.Xext:$extend)>; + } + let Predicates = !listconcat([DisallowMisalignedMemAccesses], preds) in { + def : Pat<(VecTy (load_aligned (ro.Wpat GPR64sp:$Rn, GPR32:$Rm, ro.Wext:$extend))), + (LOADW GPR64sp:$Rn, GPR32:$Rm, ro.Wext:$extend)>; - def : Pat<(VecTy (load (ro.Xpat GPR64sp:$Rn, GPR64:$Rm, ro.Xext:$extend))), - (LOADX GPR64sp:$Rn, GPR64:$Rm, ro.Xext:$extend)>; + def : Pat<(VecTy (load_aligned (ro.Xpat GPR64sp:$Rn, GPR64:$Rm, ro.Xext:$extend))), + (LOADX GPR64sp:$Rn, GPR64:$Rm, ro.Xext:$extend)>; + } } -let AddedComplexity = 10 in { -let Predicates = [IsLE] in { - // We must do vector loads with LD1 in big-endian. - defm : VecROLoadPat; - defm : VecROLoadPat; - defm : VecROLoadPat; - defm : VecROLoadPat; - defm : VecROLoadPat; - defm : VecROLoadPat; -} +// We must do vector loads with LD1 in big-endian, or when the alignment is +// less than the size of the vector. -defm : VecROLoadPat; -defm : VecROLoadPat; +// Match all load 64 bits width whose type is compatible with FPR64 +let AddedComplexity = 10 in { +defm : VecROLoadPat; +defm : VecROLoadPat; +defm : VecROLoadPat; +defm : VecROLoadPat; +defm : VecROLoadPat; +defm : VecROLoadPat; +defm : VecROLoadPat; +defm : VecROLoadPat; // Match all load 128 bits width whose type is compatible with FPR128 -let Predicates = [IsLE] in { - // We must do vector loads with LD1 in big-endian. - defm : VecROLoadPat; - defm : VecROLoadPat; - defm : VecROLoadPat; - defm : VecROLoadPat; - defm : VecROLoadPat; - defm : VecROLoadPat; - defm : VecROLoadPat; - defm : VecROLoadPat; -} +defm : VecROLoadPat; +defm : VecROLoadPat; +defm : VecROLoadPat; +defm : VecROLoadPat; +defm : VecROLoadPat; +defm : VecROLoadPat; +defm : VecROLoadPat; +defm : VecROLoadPat; } // AddedComplexity = 10 // zextload -> i64 @@ -3985,9 +4097,12 @@ defm LDRQ : LoadUI<0b00, 1, 0b11, FPR128Op, uimm12s16, "ldr", def : Pat <(bf16 (load (am_indexed16 GPR64sp:$Rn, uimm12s2:$offset))), (LDRHui GPR64sp:$Rn, uimm12s2:$offset)>; +// We must use LD1 to perform vector loads in big-endian, or when the +// alignment is less than the vector size. + +let AddedComplexity = 10 in { // Match all load 64 bits width whose type is compatible with FPR64 -let Predicates = [IsLE] in { - // We must use LD1 to perform vector loads in big-endian. +let Predicates = [IsLE, AllowMisalignedMemAccesses] in { def : Pat<(v2f32 (load (am_indexed64 GPR64sp:$Rn, uimm12s8:$offset))), (LDRDui GPR64sp:$Rn, uimm12s8:$offset)>; def : Pat<(v8i8 (load (am_indexed64 GPR64sp:$Rn, uimm12s8:$offset))), @@ -4006,9 +4121,23 @@ def : Pat<(v1f64 (load (am_indexed64 GPR64sp:$Rn, uimm12s8:$offset))), def : Pat<(v1i64 (load (am_indexed64 GPR64sp:$Rn, uimm12s8:$offset))), (LDRDui GPR64sp:$Rn, uimm12s8:$offset)>; +let Predicates = [IsLE, DisallowMisalignedMemAccesses] in { + def : Pat<(v2f32 (load_aligned<8> (am_indexed64 GPR64sp:$Rn, uimm12s8:$offset))), + (LDRDui GPR64sp:$Rn, uimm12s8:$offset)>; + def : Pat<(v8i8 (load_aligned<8> (am_indexed64 GPR64sp:$Rn, uimm12s8:$offset))), + (LDRDui GPR64sp:$Rn, uimm12s8:$offset)>; + def : Pat<(v4i16 (load_aligned<8> (am_indexed64 GPR64sp:$Rn, uimm12s8:$offset))), + (LDRDui GPR64sp:$Rn, uimm12s8:$offset)>; + def : Pat<(v2i32 (load_aligned<8> (am_indexed64 GPR64sp:$Rn, uimm12s8:$offset))), + (LDRDui GPR64sp:$Rn, uimm12s8:$offset)>; + def : Pat<(v4f16 (load_aligned<8> (am_indexed64 GPR64sp:$Rn, uimm12s8:$offset))), + (LDRDui GPR64sp:$Rn, uimm12s8:$offset)>; + def : Pat<(v4bf16 (load_aligned<8> (am_indexed64 GPR64sp:$Rn, uimm12s8:$offset))), + (LDRDui GPR64sp:$Rn, uimm12s8:$offset)>; +} + // Match all load 128 bits width whose type is compatible with FPR128 -let Predicates = [IsLE] in { - // We must use LD1 to perform vector loads in big-endian. +let Predicates = [IsLE, AllowMisalignedMemAccesses] in { def : Pat<(v4f32 (load (am_indexed128 GPR64sp:$Rn, uimm12s16:$offset))), (LDRQui GPR64sp:$Rn, uimm12s16:$offset)>; def : Pat<(v2f64 (load (am_indexed128 GPR64sp:$Rn, uimm12s16:$offset))), @@ -4029,6 +4158,26 @@ let Predicates = [IsLE] in { def : Pat<(f128 (load (am_indexed128 GPR64sp:$Rn, uimm12s16:$offset))), (LDRQui GPR64sp:$Rn, uimm12s16:$offset)>; +let Predicates = [IsLE, DisallowMisalignedMemAccesses] in { + def : Pat<(v4f32 (load_aligned<16> (am_indexed128 GPR64sp:$Rn, uimm12s16:$offset))), + (LDRQui GPR64sp:$Rn, uimm12s16:$offset)>; + def : Pat<(v2f64 (load_aligned<16> (am_indexed128 GPR64sp:$Rn, uimm12s16:$offset))), + (LDRQui GPR64sp:$Rn, uimm12s16:$offset)>; + def : Pat<(v16i8 (load_aligned<16> (am_indexed128 GPR64sp:$Rn, uimm12s16:$offset))), + (LDRQui GPR64sp:$Rn, uimm12s16:$offset)>; + def : Pat<(v8i16 (load_aligned<16> (am_indexed128 GPR64sp:$Rn, uimm12s16:$offset))), + (LDRQui GPR64sp:$Rn, uimm12s16:$offset)>; + def : Pat<(v4i32 (load_aligned<16> (am_indexed128 GPR64sp:$Rn, uimm12s16:$offset))), + (LDRQui GPR64sp:$Rn, uimm12s16:$offset)>; + def : Pat<(v2i64 (load_aligned<16> (am_indexed128 GPR64sp:$Rn, uimm12s16:$offset))), + (LDRQui GPR64sp:$Rn, uimm12s16:$offset)>; + def : Pat<(v8f16 (load_aligned<16> (am_indexed128 GPR64sp:$Rn, uimm12s16:$offset))), + (LDRQui GPR64sp:$Rn, uimm12s16:$offset)>; + def : Pat<(v8bf16 (load_aligned<16> (am_indexed128 GPR64sp:$Rn, uimm12s16:$offset))), + (LDRQui GPR64sp:$Rn, uimm12s16:$offset)>; +} +} // AddedComplexity = 10 + defm LDRHH : LoadUI<0b01, 0, 0b01, GPR32, uimm12s2, "ldrh", [(set GPR32:$Rt, (zextloadi16 (am_indexed16 GPR64sp:$Rn, @@ -4182,8 +4331,9 @@ defm LDURBB def : Pat <(bf16 (load (am_unscaled16 GPR64sp:$Rn, simm9:$offset))), (LDURHi GPR64sp:$Rn, simm9:$offset)>; +let AddedComplexity = 10 in { // Match all load 64 bits width whose type is compatible with FPR64 -let Predicates = [IsLE] in { +let Predicates = [IsLE, AllowMisalignedMemAccesses] in { def : Pat<(v2f32 (load (am_unscaled64 GPR64sp:$Rn, simm9:$offset))), (LDURDi GPR64sp:$Rn, simm9:$offset)>; def : Pat<(v2i32 (load (am_unscaled64 GPR64sp:$Rn, simm9:$offset))), @@ -4194,14 +4344,31 @@ let Predicates = [IsLE] in { (LDURDi GPR64sp:$Rn, simm9:$offset)>; def : Pat<(v4f16 (load (am_unscaled64 GPR64sp:$Rn, simm9:$offset))), (LDURDi GPR64sp:$Rn, simm9:$offset)>; + def : Pat<(v4bf16 (load (am_unscaled64 GPR64sp:$Rn, simm9:$offset))), + (LDURDi GPR64sp:$Rn, simm9:$offset)>; } def : Pat<(v1f64 (load (am_unscaled64 GPR64sp:$Rn, simm9:$offset))), (LDURDi GPR64sp:$Rn, simm9:$offset)>; def : Pat<(v1i64 (load (am_unscaled64 GPR64sp:$Rn, simm9:$offset))), (LDURDi GPR64sp:$Rn, simm9:$offset)>; +let Predicates = [IsLE, DisallowMisalignedMemAccesses] in { + def : Pat<(v2f32 (load_aligned<8> (am_unscaled64 GPR64sp:$Rn, simm9:$offset))), + (LDURDi GPR64sp:$Rn, simm9:$offset)>; + def : Pat<(v2i32 (load_aligned<8> (am_unscaled64 GPR64sp:$Rn, simm9:$offset))), + (LDURDi GPR64sp:$Rn, simm9:$offset)>; + def : Pat<(v4i16 (load_aligned<8> (am_unscaled64 GPR64sp:$Rn, simm9:$offset))), + (LDURDi GPR64sp:$Rn, simm9:$offset)>; + def : Pat<(v8i8 (load_aligned<8> (am_unscaled64 GPR64sp:$Rn, simm9:$offset))), + (LDURDi GPR64sp:$Rn, simm9:$offset)>; + def : Pat<(v4f16 (load_aligned<8> (am_unscaled64 GPR64sp:$Rn, simm9:$offset))), + (LDURDi GPR64sp:$Rn, simm9:$offset)>; + def : Pat<(v4bf16 (load_aligned<8> (am_unscaled64 GPR64sp:$Rn, simm9:$offset))), + (LDURDi GPR64sp:$Rn, simm9:$offset)>; +} + // Match all load 128 bits width whose type is compatible with FPR128 -let Predicates = [IsLE] in { +let Predicates = [IsLE, AllowMisalignedMemAccesses] in { def : Pat<(v2f64 (load (am_unscaled128 GPR64sp:$Rn, simm9:$offset))), (LDURQi GPR64sp:$Rn, simm9:$offset)>; def : Pat<(v2i64 (load (am_unscaled128 GPR64sp:$Rn, simm9:$offset))), @@ -4216,7 +4383,29 @@ let Predicates = [IsLE] in { (LDURQi GPR64sp:$Rn, simm9:$offset)>; def : Pat<(v8f16 (load (am_unscaled128 GPR64sp:$Rn, simm9:$offset))), (LDURQi GPR64sp:$Rn, simm9:$offset)>; + def : Pat<(v8bf16 (load (am_unscaled128 GPR64sp:$Rn, simm9:$offset))), + (LDURQi GPR64sp:$Rn, simm9:$offset)>; +} + +let Predicates = [IsLE, DisallowMisalignedMemAccesses] in { + def : Pat<(v2f64 (load_aligned<16> (am_unscaled128 GPR64sp:$Rn, simm9:$offset))), + (LDURQi GPR64sp:$Rn, simm9:$offset)>; + def : Pat<(v2i64 (load_aligned<16> (am_unscaled128 GPR64sp:$Rn, simm9:$offset))), + (LDURQi GPR64sp:$Rn, simm9:$offset)>; + def : Pat<(v4f32 (load_aligned<16> (am_unscaled128 GPR64sp:$Rn, simm9:$offset))), + (LDURQi GPR64sp:$Rn, simm9:$offset)>; + def : Pat<(v4i32 (load_aligned<16> (am_unscaled128 GPR64sp:$Rn, simm9:$offset))), + (LDURQi GPR64sp:$Rn, simm9:$offset)>; + def : Pat<(v8i16 (load_aligned<16> (am_unscaled128 GPR64sp:$Rn, simm9:$offset))), + (LDURQi GPR64sp:$Rn, simm9:$offset)>; + def : Pat<(v16i8 (load_aligned<16> (am_unscaled128 GPR64sp:$Rn, simm9:$offset))), + (LDURQi GPR64sp:$Rn, simm9:$offset)>; + def : Pat<(v8f16 (load_aligned<16> (am_unscaled128 GPR64sp:$Rn, simm9:$offset))), + (LDURQi GPR64sp:$Rn, simm9:$offset)>; + def : Pat<(v8bf16 (load_aligned<16> (am_unscaled128 GPR64sp:$Rn, simm9:$offset))), + (LDURQi GPR64sp:$Rn, simm9:$offset)>; } +} // AddedComplexity = 10 // anyext -> zext def : Pat<(i32 (extloadi16 (am_unscaled16 GPR64sp:$Rn, simm9:$offset))), @@ -4723,44 +4912,54 @@ let AddedComplexity = 10 in { defm : TruncStoreFrom64ROPat; } + multiclass VecROStorePat { - def : Pat<(store (VecTy FPR:$Rt), - (ro.Wpat GPR64sp:$Rn, GPR32:$Rm, ro.Wext:$extend)), - (STRW FPR:$Rt, GPR64sp:$Rn, GPR32:$Rm, ro.Wext:$extend)>; + Instruction STRW, Instruction STRX, int align_bytes, + list preds> { + let Predicates = !listconcat([AllowMisalignedMemAccesses], preds) in { + def : Pat<(store (VecTy FPR:$Rt), + (ro.Wpat GPR64sp:$Rn, GPR32:$Rm, ro.Wext:$extend)), + (STRW FPR:$Rt, GPR64sp:$Rn, GPR32:$Rm, ro.Wext:$extend)>; + + def : Pat<(store (VecTy FPR:$Rt), + (ro.Xpat GPR64sp:$Rn, GPR64:$Rm, ro.Xext:$extend)), + (STRX FPR:$Rt, GPR64sp:$Rn, GPR64:$Rm, ro.Xext:$extend)>; + } + + let Predicates = !listconcat([DisallowMisalignedMemAccesses], preds) in { + def : Pat<(store_aligned (VecTy FPR:$Rt), + (ro.Wpat GPR64sp:$Rn, GPR32:$Rm, ro.Wext:$extend)), + (STRW FPR:$Rt, GPR64sp:$Rn, GPR32:$Rm, ro.Wext:$extend)>; - def : Pat<(store (VecTy FPR:$Rt), - (ro.Xpat GPR64sp:$Rn, GPR64:$Rm, ro.Xext:$extend)), - (STRX FPR:$Rt, GPR64sp:$Rn, GPR64:$Rm, ro.Xext:$extend)>; + def : Pat<(store_aligned (VecTy FPR:$Rt), + (ro.Xpat GPR64sp:$Rn, GPR64:$Rm, ro.Xext:$extend)), + (STRX FPR:$Rt, GPR64sp:$Rn, GPR64:$Rm, ro.Xext:$extend)>; + } } +// We must use ST1 to store vectors in big-endian, or when the alignment is +// less than the size of the vector. let AddedComplexity = 10 in { // Match all store 64 bits width whose type is compatible with FPR64 -let Predicates = [IsLE] in { - // We must use ST1 to store vectors in big-endian. - defm : VecROStorePat; - defm : VecROStorePat; - defm : VecROStorePat; - defm : VecROStorePat; - defm : VecROStorePat; - defm : VecROStorePat; -} +defm : VecROStorePat; +defm : VecROStorePat; +defm : VecROStorePat; +defm : VecROStorePat; +defm : VecROStorePat; +defm : VecROStorePat; -defm : VecROStorePat; -defm : VecROStorePat; +defm : VecROStorePat; +defm : VecROStorePat; // Match all store 128 bits width whose type is compatible with FPR128 -let Predicates = [IsLE, UseSTRQro] in { - // We must use ST1 to store vectors in big-endian. - defm : VecROStorePat; - defm : VecROStorePat; - defm : VecROStorePat; - defm : VecROStorePat; - defm : VecROStorePat; - defm : VecROStorePat; - defm : VecROStorePat; - defm : VecROStorePat; -} +defm : VecROStorePat; +defm : VecROStorePat; +defm : VecROStorePat; +defm : VecROStorePat; +defm : VecROStorePat; +defm : VecROStorePat; +defm : VecROStorePat; +defm : VecROStorePat; } // AddedComplexity = 10 // Match stores from lane 0 to the appropriate subreg's store. @@ -4830,6 +5029,10 @@ def : Pat<(store (bf16 FPR16Op:$Rt), let AddedComplexity = 10 in { +// We must use ST1 to store vectors in big-endian, or when the aligmnment is +// less than the size of the vector. + +let AddedComplexity = 10 in { // Match all store 64 bits width whose type is compatible with FPR64 def : Pat<(store (v1i64 FPR64:$Rt), (am_indexed64 GPR64sp:$Rn, uimm12s8:$offset)), @@ -4838,8 +5041,7 @@ def : Pat<(store (v1f64 FPR64:$Rt), (am_indexed64 GPR64sp:$Rn, uimm12s8:$offset)), (STRDui FPR64:$Rt, GPR64sp:$Rn, uimm12s8:$offset)>; -let Predicates = [IsLE] in { - // We must use ST1 to store vectors in big-endian. +let Predicates = [IsLE, AllowMisalignedMemAccesses] in { def : Pat<(store (v2f32 FPR64:$Rt), (am_indexed64 GPR64sp:$Rn, uimm12s8:$offset)), (STRDui FPR64:$Rt, GPR64sp:$Rn, uimm12s8:$offset)>; @@ -4860,13 +5062,33 @@ let Predicates = [IsLE] in { (STRDui FPR64:$Rt, GPR64sp:$Rn, uimm12s8:$offset)>; } +let Predicates = [IsLE, DisallowMisalignedMemAccesses] in { + def : Pat<(store_aligned<8> (v2f32 FPR64:$Rt), + (am_indexed64 GPR64sp:$Rn, uimm12s8:$offset)), + (STRDui FPR64:$Rt, GPR64sp:$Rn, uimm12s8:$offset)>; + def : Pat<(store_aligned<8> (v8i8 FPR64:$Rt), + (am_indexed64 GPR64sp:$Rn, uimm12s8:$offset)), + (STRDui FPR64:$Rt, GPR64sp:$Rn, uimm12s8:$offset)>; + def : Pat<(store_aligned<8> (v4i16 FPR64:$Rt), + (am_indexed64 GPR64sp:$Rn, uimm12s8:$offset)), + (STRDui FPR64:$Rt, GPR64sp:$Rn, uimm12s8:$offset)>; + def : Pat<(store_aligned<8> (v2i32 FPR64:$Rt), + (am_indexed64 GPR64sp:$Rn, uimm12s8:$offset)), + (STRDui FPR64:$Rt, GPR64sp:$Rn, uimm12s8:$offset)>; + def : Pat<(store_aligned<8> (v4f16 FPR64:$Rt), + (am_indexed64 GPR64sp:$Rn, uimm12s8:$offset)), + (STRDui FPR64:$Rt, GPR64sp:$Rn, uimm12s8:$offset)>; + def : Pat<(store_aligned<8> (v4bf16 FPR64:$Rt), + (am_indexed64 GPR64sp:$Rn, uimm12s8:$offset)), + (STRDui FPR64:$Rt, GPR64sp:$Rn, uimm12s8:$offset)>; +} + // Match all store 128 bits width whose type is compatible with FPR128 def : Pat<(store (f128 FPR128:$Rt), (am_indexed128 GPR64sp:$Rn, uimm12s16:$offset)), (STRQui FPR128:$Rt, GPR64sp:$Rn, uimm12s16:$offset)>; -let Predicates = [IsLE] in { - // We must use ST1 to store vectors in big-endian. +let Predicates = [IsLE, AllowMisalignedMemAccesses] in { def : Pat<(store (v4f32 FPR128:$Rt), (am_indexed128 GPR64sp:$Rn, uimm12s16:$offset)), (STRQui FPR128:$Rt, GPR64sp:$Rn, uimm12s16:$offset)>; @@ -4893,6 +5115,34 @@ let Predicates = [IsLE] in { (STRQui FPR128:$Rt, GPR64sp:$Rn, uimm12s16:$offset)>; } +let Predicates = [IsLE, DisallowMisalignedMemAccesses] in { + def : Pat<(store_aligned<16> (v4f32 FPR128:$Rt), + (am_indexed128 GPR64sp:$Rn, uimm12s16:$offset)), + (STRQui FPR128:$Rt, GPR64sp:$Rn, uimm12s16:$offset)>; + def : Pat<(store_aligned<16> (v2f64 FPR128:$Rt), + (am_indexed128 GPR64sp:$Rn, uimm12s16:$offset)), + (STRQui FPR128:$Rt, GPR64sp:$Rn, uimm12s16:$offset)>; + def : Pat<(store_aligned<16> (v16i8 FPR128:$Rt), + (am_indexed128 GPR64sp:$Rn, uimm12s16:$offset)), + (STRQui FPR128:$Rt, GPR64sp:$Rn, uimm12s16:$offset)>; + def : Pat<(store_aligned<16> (v8i16 FPR128:$Rt), + (am_indexed128 GPR64sp:$Rn, uimm12s16:$offset)), + (STRQui FPR128:$Rt, GPR64sp:$Rn, uimm12s16:$offset)>; + def : Pat<(store_aligned<16> (v4i32 FPR128:$Rt), + (am_indexed128 GPR64sp:$Rn, uimm12s16:$offset)), + (STRQui FPR128:$Rt, GPR64sp:$Rn, uimm12s16:$offset)>; + def : Pat<(store_aligned<16> (v2i64 FPR128:$Rt), + (am_indexed128 GPR64sp:$Rn, uimm12s16:$offset)), + (STRQui FPR128:$Rt, GPR64sp:$Rn, uimm12s16:$offset)>; + def : Pat<(store_aligned<16> (v8f16 FPR128:$Rt), + (am_indexed128 GPR64sp:$Rn, uimm12s16:$offset)), + (STRQui FPR128:$Rt, GPR64sp:$Rn, uimm12s16:$offset)>; + def : Pat<(store_aligned<16> (v8bf16 FPR128:$Rt), + (am_indexed128 GPR64sp:$Rn, uimm12s16:$offset)), + (STRQui FPR128:$Rt, GPR64sp:$Rn, uimm12s16:$offset)>; +} +} // AddedComplexity = 10 + // truncstorei32 of f64 bitcasted to i64 def : Pat<(truncstorei32 (i64 (bitconvert (f64 FPR64:$Rt))), (am_indexed32 GPR64sp:$Rn, uimm12s4:$offset)), (STRSui (EXTRACT_SUBREG FPR64:$Rt, ssub), GPR64sp:$Rn, uimm12s4:$offset)>; @@ -5009,8 +5259,9 @@ def : Pat<(store (v1i64 FPR64:$Rt), (am_unscaled64 GPR64sp:$Rn, simm9:$offset)), let AddedComplexity = 10 in { -let Predicates = [IsLE] in { - // We must use ST1 to store vectors in big-endian. +// We must use ST1 to store vectors in big-endian, or when the alignment is +// less than the size of the vector. +let Predicates = [IsLE, AllowMisalignedMemAccesses] in { def : Pat<(store (v2f32 FPR64:$Rt), (am_unscaled64 GPR64sp:$Rn, simm9:$offset)), (STURDi FPR64:$Rt, GPR64sp:$Rn, simm9:$offset)>; @@ -5031,12 +5282,34 @@ let Predicates = [IsLE] in { (STURDi FPR64:$Rt, GPR64sp:$Rn, simm9:$offset)>; } +let Predicates = [IsLE, DisallowMisalignedMemAccesses] in { + def : Pat<(store_aligned<8> (v2f32 FPR64:$Rt), + (am_unscaled64 GPR64sp:$Rn, simm9:$offset)), + (STURDi FPR64:$Rt, GPR64sp:$Rn, simm9:$offset)>; + def : Pat<(store_aligned<8> (v8i8 FPR64:$Rt), + (am_unscaled64 GPR64sp:$Rn, simm9:$offset)), + (STURDi FPR64:$Rt, GPR64sp:$Rn, simm9:$offset)>; + def : Pat<(store_aligned<8> (v4i16 FPR64:$Rt), + (am_unscaled64 GPR64sp:$Rn, simm9:$offset)), + (STURDi FPR64:$Rt, GPR64sp:$Rn, simm9:$offset)>; + def : Pat<(store_aligned<8> (v2i32 FPR64:$Rt), + (am_unscaled64 GPR64sp:$Rn, simm9:$offset)), + (STURDi FPR64:$Rt, GPR64sp:$Rn, simm9:$offset)>; + def : Pat<(store_aligned<8> (v4f16 FPR64:$Rt), + (am_unscaled64 GPR64sp:$Rn, simm9:$offset)), + (STURDi FPR64:$Rt, GPR64sp:$Rn, simm9:$offset)>; + def : Pat<(store_aligned<8> (v4bf16 FPR64:$Rt), + (am_unscaled64 GPR64sp:$Rn, simm9:$offset)), + (STURDi FPR64:$Rt, GPR64sp:$Rn, simm9:$offset)>; +} + // Match all store 128 bits width whose type is compatible with FPR128 def : Pat<(store (f128 FPR128:$Rt), (am_unscaled128 GPR64sp:$Rn, simm9:$offset)), (STURQi FPR128:$Rt, GPR64sp:$Rn, simm9:$offset)>; -let Predicates = [IsLE] in { - // We must use ST1 to store vectors in big-endian. +// We must use ST1 to store vectors in big-endian, or when the alignment is +// less than the size of the vector. +let Predicates = [IsLE, AllowMisalignedMemAccesses] in { def : Pat<(store (v4f32 FPR128:$Rt), (am_unscaled128 GPR64sp:$Rn, simm9:$offset)), (STURQi FPR128:$Rt, GPR64sp:$Rn, simm9:$offset)>; @@ -5066,6 +5339,36 @@ let Predicates = [IsLE] in { (STURQi FPR128:$Rt, GPR64sp:$Rn, simm9:$offset)>; } +let Predicates = [IsLE, DisallowMisalignedMemAccesses] in { + def : Pat<(store_aligned<16> (v4f32 FPR128:$Rt), + (am_unscaled128 GPR64sp:$Rn, simm9:$offset)), + (STURQi FPR128:$Rt, GPR64sp:$Rn, simm9:$offset)>; + def : Pat<(store_aligned<16> (v2f64 FPR128:$Rt), + (am_unscaled128 GPR64sp:$Rn, simm9:$offset)), + (STURQi FPR128:$Rt, GPR64sp:$Rn, simm9:$offset)>; + def : Pat<(store_aligned<16> (v16i8 FPR128:$Rt), + (am_unscaled128 GPR64sp:$Rn, simm9:$offset)), + (STURQi FPR128:$Rt, GPR64sp:$Rn, simm9:$offset)>; + def : Pat<(store_aligned<16> (v8i16 FPR128:$Rt), + (am_unscaled128 GPR64sp:$Rn, simm9:$offset)), + (STURQi FPR128:$Rt, GPR64sp:$Rn, simm9:$offset)>; + def : Pat<(store_aligned<16> (v4i32 FPR128:$Rt), + (am_unscaled128 GPR64sp:$Rn, simm9:$offset)), + (STURQi FPR128:$Rt, GPR64sp:$Rn, simm9:$offset)>; + def : Pat<(store_aligned<16> (v2i64 FPR128:$Rt), + (am_unscaled128 GPR64sp:$Rn, simm9:$offset)), + (STURQi FPR128:$Rt, GPR64sp:$Rn, simm9:$offset)>; + def : Pat<(store_aligned<16> (v2f64 FPR128:$Rt), + (am_unscaled128 GPR64sp:$Rn, simm9:$offset)), + (STURQi FPR128:$Rt, GPR64sp:$Rn, simm9:$offset)>; + def : Pat<(store_aligned<16> (v8f16 FPR128:$Rt), + (am_unscaled128 GPR64sp:$Rn, simm9:$offset)), + (STURQi FPR128:$Rt, GPR64sp:$Rn, simm9:$offset)>; + def : Pat<(store_aligned<16> (v8bf16 FPR128:$Rt), + (am_unscaled128 GPR64sp:$Rn, simm9:$offset)), + (STURQi FPR128:$Rt, GPR64sp:$Rn, simm9:$offset)>; +} + } // AddedComplexity = 10 // unscaled i64 truncating stores @@ -5161,39 +5464,77 @@ def : Pat<(pre_truncsti8 GPR64:$Rt, GPR64sp:$addr, simm9:$off), (STRBBpre (EXTRACT_SUBREG GPR64:$Rt, sub_32), GPR64sp:$addr, simm9:$off)>; -def : Pat<(pre_store (v8i8 FPR64:$Rt), GPR64sp:$addr, simm9:$off), - (STRDpre FPR64:$Rt, GPR64sp:$addr, simm9:$off)>; -def : Pat<(pre_store (v4i16 FPR64:$Rt), GPR64sp:$addr, simm9:$off), - (STRDpre FPR64:$Rt, GPR64sp:$addr, simm9:$off)>; -def : Pat<(pre_store (v2i32 FPR64:$Rt), GPR64sp:$addr, simm9:$off), - (STRDpre FPR64:$Rt, GPR64sp:$addr, simm9:$off)>; -def : Pat<(pre_store (v2f32 FPR64:$Rt), GPR64sp:$addr, simm9:$off), - (STRDpre FPR64:$Rt, GPR64sp:$addr, simm9:$off)>; -def : Pat<(pre_store (v1i64 FPR64:$Rt), GPR64sp:$addr, simm9:$off), - (STRDpre FPR64:$Rt, GPR64sp:$addr, simm9:$off)>; -def : Pat<(pre_store (v1f64 FPR64:$Rt), GPR64sp:$addr, simm9:$off), - (STRDpre FPR64:$Rt, GPR64sp:$addr, simm9:$off)>; -def : Pat<(pre_store (v4f16 FPR64:$Rt), GPR64sp:$addr, simm9:$off), - (STRDpre FPR64:$Rt, GPR64sp:$addr, simm9:$off)>; -def : Pat<(pre_store (v4bf16 FPR64:$Rt), GPR64sp:$addr, simm9:$off), - (STRDpre FPR64:$Rt, GPR64sp:$addr, simm9:$off)>; - -def : Pat<(pre_store (v16i8 FPR128:$Rt), GPR64sp:$addr, simm9:$off), - (STRQpre FPR128:$Rt, GPR64sp:$addr, simm9:$off)>; -def : Pat<(pre_store (v8i16 FPR128:$Rt), GPR64sp:$addr, simm9:$off), - (STRQpre FPR128:$Rt, GPR64sp:$addr, simm9:$off)>; -def : Pat<(pre_store (v4i32 FPR128:$Rt), GPR64sp:$addr, simm9:$off), - (STRQpre FPR128:$Rt, GPR64sp:$addr, simm9:$off)>; -def : Pat<(pre_store (v4f32 FPR128:$Rt), GPR64sp:$addr, simm9:$off), - (STRQpre FPR128:$Rt, GPR64sp:$addr, simm9:$off)>; -def : Pat<(pre_store (v2i64 FPR128:$Rt), GPR64sp:$addr, simm9:$off), - (STRQpre FPR128:$Rt, GPR64sp:$addr, simm9:$off)>; -def : Pat<(pre_store (v2f64 FPR128:$Rt), GPR64sp:$addr, simm9:$off), - (STRQpre FPR128:$Rt, GPR64sp:$addr, simm9:$off)>; -def : Pat<(pre_store (v8f16 FPR128:$Rt), GPR64sp:$addr, simm9:$off), - (STRQpre FPR128:$Rt, GPR64sp:$addr, simm9:$off)>; -def : Pat<(pre_store (v8bf16 FPR128:$Rt), GPR64sp:$addr, simm9:$off), - (STRQpre FPR128:$Rt, GPR64sp:$addr, simm9:$off)>; +let Predicates = [AllowMisalignedMemAccesses] in { + def : Pat<(pre_store (v8i8 FPR64:$Rt), GPR64sp:$addr, simm9:$off), + (STRDpre FPR64:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(pre_store (v4i16 FPR64:$Rt), GPR64sp:$addr, simm9:$off), + (STRDpre FPR64:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(pre_store (v2i32 FPR64:$Rt), GPR64sp:$addr, simm9:$off), + (STRDpre FPR64:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(pre_store (v2f32 FPR64:$Rt), GPR64sp:$addr, simm9:$off), + (STRDpre FPR64:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(pre_store (v1i64 FPR64:$Rt), GPR64sp:$addr, simm9:$off), + (STRDpre FPR64:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(pre_store (v1f64 FPR64:$Rt), GPR64sp:$addr, simm9:$off), + (STRDpre FPR64:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(pre_store (v4f16 FPR64:$Rt), GPR64sp:$addr, simm9:$off), + (STRDpre FPR64:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(pre_store (v4bf16 FPR64:$Rt), GPR64sp:$addr, simm9:$off), + (STRDpre FPR64:$Rt, GPR64sp:$addr, simm9:$off)>; + + def : Pat<(pre_store (v16i8 FPR128:$Rt), GPR64sp:$addr, simm9:$off), + (STRQpre FPR128:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(pre_store (v8i16 FPR128:$Rt), GPR64sp:$addr, simm9:$off), + (STRQpre FPR128:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(pre_store (v4i32 FPR128:$Rt), GPR64sp:$addr, simm9:$off), + (STRQpre FPR128:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(pre_store (v4f32 FPR128:$Rt), GPR64sp:$addr, simm9:$off), + (STRQpre FPR128:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(pre_store (v2i64 FPR128:$Rt), GPR64sp:$addr, simm9:$off), + (STRQpre FPR128:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(pre_store (v2f64 FPR128:$Rt), GPR64sp:$addr, simm9:$off), + (STRQpre FPR128:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(pre_store (v8f16 FPR128:$Rt), GPR64sp:$addr, simm9:$off), + (STRQpre FPR128:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(pre_store (v8bf16 FPR128:$Rt), GPR64sp:$addr, simm9:$off), + (STRQpre FPR128:$Rt, GPR64sp:$addr, simm9:$off)>; +} + +let Predicates = [DisallowMisalignedMemAccesses] in { + def : Pat<(pre_store_aligned<8> (v8i8 FPR64:$Rt), GPR64sp:$addr, simm9:$off), + (STRDpre FPR64:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(pre_store_aligned<8> (v4i16 FPR64:$Rt), GPR64sp:$addr, simm9:$off), + (STRDpre FPR64:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(pre_store_aligned<8> (v2i32 FPR64:$Rt), GPR64sp:$addr, simm9:$off), + (STRDpre FPR64:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(pre_store_aligned<8> (v2f32 FPR64:$Rt), GPR64sp:$addr, simm9:$off), + (STRDpre FPR64:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(pre_store_aligned<8> (v1i64 FPR64:$Rt), GPR64sp:$addr, simm9:$off), + (STRDpre FPR64:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(pre_store_aligned<8> (v1f64 FPR64:$Rt), GPR64sp:$addr, simm9:$off), + (STRDpre FPR64:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(pre_store_aligned<8> (v4f16 FPR64:$Rt), GPR64sp:$addr, simm9:$off), + (STRDpre FPR64:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(pre_store_aligned<8> (v4bf16 FPR64:$Rt), GPR64sp:$addr, simm9:$off), + (STRDpre FPR64:$Rt, GPR64sp:$addr, simm9:$off)>; + + def : Pat<(pre_store_aligned<16> (v16i8 FPR128:$Rt), GPR64sp:$addr, simm9:$off), + (STRQpre FPR128:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(pre_store_aligned<16> (v8i16 FPR128:$Rt), GPR64sp:$addr, simm9:$off), + (STRQpre FPR128:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(pre_store_aligned<16> (v4i32 FPR128:$Rt), GPR64sp:$addr, simm9:$off), + (STRQpre FPR128:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(pre_store_aligned<16> (v4f32 FPR128:$Rt), GPR64sp:$addr, simm9:$off), + (STRQpre FPR128:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(pre_store_aligned<16> (v2i64 FPR128:$Rt), GPR64sp:$addr, simm9:$off), + (STRQpre FPR128:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(pre_store_aligned<16> (v2f64 FPR128:$Rt), GPR64sp:$addr, simm9:$off), + (STRQpre FPR128:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(pre_store_aligned<16> (v8f16 FPR128:$Rt), GPR64sp:$addr, simm9:$off), + (STRQpre FPR128:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(pre_store_aligned<16> (v8bf16 FPR128:$Rt), GPR64sp:$addr, simm9:$off), + (STRQpre FPR128:$Rt, GPR64sp:$addr, simm9:$off)>; +} //--- // (immediate post-indexed) @@ -5224,7 +5565,8 @@ def : Pat<(post_truncsti8 GPR64:$Rt, GPR64sp:$addr, simm9:$off), def : Pat<(post_store (bf16 FPR16:$Rt), GPR64sp:$addr, simm9:$off), (STRHpost FPR16:$Rt, GPR64sp:$addr, simm9:$off)>; -let Predicates = [IsLE] in { +let AddedComplexity = 10 in { +let Predicates = [IsLE, AllowMisalignedMemAccesses] in { // We must use ST1 to store vectors in big-endian. def : Pat<(post_store(v8i8 FPR64:$Rt), GPR64sp:$addr, simm9:$off), (STRDpost FPR64:$Rt, GPR64sp:$addr, simm9:$off)>; @@ -5261,6 +5603,44 @@ let Predicates = [IsLE] in { (STRQpost FPR128:$Rt, GPR64sp:$addr, simm9:$off)>; } +let Predicates = [IsLE, DisallowMisalignedMemAccesses] in { + // We must use ST1 to store vectors in big-endian. + def : Pat<(post_store_aligned<8> (v8i8 FPR64:$Rt), GPR64sp:$addr, simm9:$off), + (STRDpost FPR64:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(post_store_aligned<8> (v4i16 FPR64:$Rt), GPR64sp:$addr, simm9:$off), + (STRDpost FPR64:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(post_store_aligned<8> (v2i32 FPR64:$Rt), GPR64sp:$addr, simm9:$off), + (STRDpost FPR64:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(post_store_aligned<8> (v2f32 FPR64:$Rt), GPR64sp:$addr, simm9:$off), + (STRDpost FPR64:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(post_store_aligned<8> (v1i64 FPR64:$Rt), GPR64sp:$addr, simm9:$off), + (STRDpost FPR64:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(post_store_aligned<8> (v1f64 FPR64:$Rt), GPR64sp:$addr, simm9:$off), + (STRDpost FPR64:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(post_store_aligned<8> (v4f16 FPR64:$Rt), GPR64sp:$addr, simm9:$off), + (STRDpost FPR64:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(post_store_aligned<8> (v4bf16 FPR64:$Rt), GPR64sp:$addr, simm9:$off), + (STRDpost FPR64:$Rt, GPR64sp:$addr, simm9:$off)>; + + def : Pat<(post_store_aligned<16> (v16i8 FPR128:$Rt), GPR64sp:$addr, simm9:$off), + (STRQpost FPR128:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(post_store_aligned<16> (v8i16 FPR128:$Rt), GPR64sp:$addr, simm9:$off), + (STRQpost FPR128:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(post_store_aligned<16> (v4i32 FPR128:$Rt), GPR64sp:$addr, simm9:$off), + (STRQpost FPR128:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(post_store_aligned<16> (v4f32 FPR128:$Rt), GPR64sp:$addr, simm9:$off), + (STRQpost FPR128:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(post_store_aligned<16> (v2i64 FPR128:$Rt), GPR64sp:$addr, simm9:$off), + (STRQpost FPR128:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(post_store_aligned<16> (v2f64 FPR128:$Rt), GPR64sp:$addr, simm9:$off), + (STRQpost FPR128:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(post_store_aligned<16> (v8f16 FPR128:$Rt), GPR64sp:$addr, simm9:$off), + (STRQpost FPR128:$Rt, GPR64sp:$addr, simm9:$off)>; + def : Pat<(post_store_aligned<16> (v8bf16 FPR128:$Rt), GPR64sp:$addr, simm9:$off), + (STRQpost FPR128:$Rt, GPR64sp:$addr, simm9:$off)>; +} +} // AddedComplexity = 10 + //===----------------------------------------------------------------------===// // Load/store exclusive instructions. //===----------------------------------------------------------------------===// @@ -5560,6 +5940,13 @@ def : Pat<(i64 (any_llrint f32:$Rn)), def : Pat<(i64 (any_llrint f64:$Rn)), (FCVTZSUXDr (FRINTXDr f64:$Rn))>; +let Predicates = [HasFullFP16] in { + def : Pat<(bf16 (fabs (bf16 FPR16:$Rn))), + (bf16 (FABSHr (bf16 FPR16:$Rn)))>; + def : Pat<(bf16 (fneg (bf16 FPR16:$Rn))), + (bf16 (FNEGHr (bf16 FPR16:$Rn)))>; +} + //===----------------------------------------------------------------------===// // Floating point two operand instructions. //===----------------------------------------------------------------------===// @@ -5824,11 +6211,6 @@ defm FCMLT : SIMDFPCmpTwoVector<0, 1, 0b01110, "fcmlt", AArch64fcmltz>; defm FCVTAS : SIMDTwoVectorFPToInt<0,0,0b11100, "fcvtas",int_aarch64_neon_fcvtas>; defm FCVTAU : SIMDTwoVectorFPToInt<1,0,0b11100, "fcvtau",int_aarch64_neon_fcvtau>; defm FCVTL : SIMDFPWidenTwoVector<0, 0, 0b10111, "fcvtl">; -def : Pat<(v4f32 (int_aarch64_neon_vcvthf2fp (v4i16 V64:$Rn))), - (FCVTLv4i16 V64:$Rn)>; -def : Pat<(v4f32 (int_aarch64_neon_vcvthf2fp (extract_subvector (v8i16 V128:$Rn), - (i64 4)))), - (FCVTLv8i16 V128:$Rn)>; def : Pat<(v2f64 (any_fpextend (v2f32 V64:$Rn))), (FCVTLv2i32 V64:$Rn)>; def : Pat<(v2f64 (any_fpextend (v2f32 (extract_high_v4f32 (v4f32 V128:$Rn))))), @@ -5842,11 +6224,6 @@ defm FCVTMU : SIMDTwoVectorFPToInt<1,0,0b11011, "fcvtmu",int_aarch64_neon_fcvtmu defm FCVTNS : SIMDTwoVectorFPToInt<0,0,0b11010, "fcvtns",int_aarch64_neon_fcvtns>; defm FCVTNU : SIMDTwoVectorFPToInt<1,0,0b11010, "fcvtnu",int_aarch64_neon_fcvtnu>; defm FCVTN : SIMDFPNarrowTwoVector<0, 0, 0b10110, "fcvtn">; -def : Pat<(v4i16 (int_aarch64_neon_vcvtfp2hf (v4f32 V128:$Rn))), - (FCVTNv4i16 V128:$Rn)>; -def : Pat<(concat_vectors V64:$Rd, - (v4i16 (int_aarch64_neon_vcvtfp2hf (v4f32 V128:$Rn)))), - (FCVTNv8i16 (INSERT_SUBREG (IMPLICIT_DEF), V64:$Rd, dsub), V128:$Rn)>; def : Pat<(v2f32 (any_fpround (v2f64 V128:$Rn))), (FCVTNv2i32 V128:$Rn)>; def : Pat<(v4f16 (any_fpround (v4f32 V128:$Rn))), @@ -5862,17 +6239,12 @@ defm FCVTXN : SIMDFPInexactCvtTwoVector<1, 0, 0b10110, "fcvtxn", defm FCVTZS : SIMDTwoVectorFPToInt<0, 1, 0b11011, "fcvtzs", any_fp_to_sint>; defm FCVTZU : SIMDTwoVectorFPToInt<1, 1, 0b11011, "fcvtzu", any_fp_to_uint>; // AArch64's FCVT instructions saturate when out of range. -multiclass SIMDTwoVectorFPToIntSatPats { +multiclass SIMDTwoVectorFPToIntSatPats { let Predicates = [HasNEONandIsFPRCVTStreamingSafe, HasFullFP16] in { def : Pat<(v4i16 (to_int_sat v4f16:$Rn, i16)), (!cast(INST # v4f16) v4f16:$Rn)>; def : Pat<(v8i16 (to_int_sat v8f16:$Rn, i16)), (!cast(INST # v8f16) v8f16:$Rn)>; - - def : Pat<(v4i16 (to_int_sat_gi v4f16:$Rn)), - (!cast(INST # v4f16) v4f16:$Rn)>; - def : Pat<(v8i16 (to_int_sat_gi v8f16:$Rn)), - (!cast(INST # v8f16) v8f16:$Rn)>; } let Predicates = [HasNEONandIsFPRCVTStreamingSafe] in { def : Pat<(v2i32 (to_int_sat v2f32:$Rn, i32)), @@ -5881,21 +6253,14 @@ multiclass SIMDTwoVectorFPToIntSatPats(INST # v4f32) v4f32:$Rn)>; def : Pat<(v2i64 (to_int_sat v2f64:$Rn, i64)), (!cast(INST # v2f64) v2f64:$Rn)>; - - def : Pat<(v2i32 (to_int_sat_gi v2f32:$Rn)), - (!cast(INST # v2f32) v2f32:$Rn)>; - def : Pat<(v4i32 (to_int_sat_gi v4f32:$Rn)), - (!cast(INST # v4f32) v4f32:$Rn)>; - def : Pat<(v2i64 (to_int_sat_gi v2f64:$Rn)), - (!cast(INST # v2f64) v2f64:$Rn)>; } } -defm : SIMDTwoVectorFPToIntSatPats; -defm : SIMDTwoVectorFPToIntSatPats; +defm : SIMDTwoVectorFPToIntSatPats; +defm : SIMDTwoVectorFPToIntSatPats; // Fused round + convert to int patterns for vectors -multiclass SIMDTwoVectorFPToIntRoundPats { +multiclass SIMDTwoVectorFPToIntRoundPats { let Predicates = [HasFullFP16] in { def : Pat<(v4i16 (to_int (round v4f16:$Rn))), (!cast(INST # v4f16) v4f16:$Rn)>; @@ -5906,11 +6271,6 @@ multiclass SIMDTwoVectorFPToIntRoundPats(INST # v4f16) v4f16:$Rn)>; def : Pat<(v8i16 (to_int_sat (round v8f16:$Rn), i16)), (!cast(INST # v8f16) v8f16:$Rn)>; - - def : Pat<(v4i16 (to_int_sat_gi (round v4f16:$Rn))), - (!cast(INST # v4f16) v4f16:$Rn)>; - def : Pat<(v8i16 (to_int_sat_gi (round v8f16:$Rn))), - (!cast(INST # v8f16) v8f16:$Rn)>; } def : Pat<(v2i32 (to_int (round v2f32:$Rn))), (!cast(INST # v2f32) v2f32:$Rn)>; @@ -5925,25 +6285,18 @@ multiclass SIMDTwoVectorFPToIntRoundPats(INST # v4f32) v4f32:$Rn)>; def : Pat<(v2i64 (to_int_sat (round v2f64:$Rn), i64)), (!cast(INST # v2f64) v2f64:$Rn)>; - - def : Pat<(v2i32 (to_int_sat_gi (round v2f32:$Rn))), - (!cast(INST # v2f32) v2f32:$Rn)>; - def : Pat<(v4i32 (to_int_sat_gi (round v4f32:$Rn))), - (!cast(INST # v4f32) v4f32:$Rn)>; - def : Pat<(v2i64 (to_int_sat_gi (round v2f64:$Rn))), - (!cast(INST # v2f64) v2f64:$Rn)>; } -defm : SIMDTwoVectorFPToIntRoundPats; -defm : SIMDTwoVectorFPToIntRoundPats; -defm : SIMDTwoVectorFPToIntRoundPats; -defm : SIMDTwoVectorFPToIntRoundPats; -defm : SIMDTwoVectorFPToIntRoundPats; -defm : SIMDTwoVectorFPToIntRoundPats; -defm : SIMDTwoVectorFPToIntRoundPats; -defm : SIMDTwoVectorFPToIntRoundPats; -defm : SIMDTwoVectorFPToIntRoundPats; -defm : SIMDTwoVectorFPToIntRoundPats; +defm : SIMDTwoVectorFPToIntRoundPats; +defm : SIMDTwoVectorFPToIntRoundPats; +defm : SIMDTwoVectorFPToIntRoundPats; +defm : SIMDTwoVectorFPToIntRoundPats; +defm : SIMDTwoVectorFPToIntRoundPats; +defm : SIMDTwoVectorFPToIntRoundPats; +defm : SIMDTwoVectorFPToIntRoundPats; +defm : SIMDTwoVectorFPToIntRoundPats; +defm : SIMDTwoVectorFPToIntRoundPats; +defm : SIMDTwoVectorFPToIntRoundPats; let Predicates = [HasNEONandIsFPRCVTStreamingSafe] in { def : Pat<(v4i16 (int_aarch64_neon_fcvtzs v4f16:$Rn)), (FCVTZSv4f16 $Rn)>; @@ -5989,6 +6342,17 @@ def : InstAlias<"mvn{ $Vd.16b, $Vn.16b|.16b $Vd, $Vn}", (NOTv16i8 V128:$Vd, V128:$Vn)>; } +let Predicates = [HasFullFP16] in { + def : Pat<(v4bf16 (fabs (v4bf16 V64:$Rn))), + (v4bf16 (FABSv4f16 (v4bf16 V64:$Rn)))>; + def : Pat<(v8bf16 (fabs (v8bf16 V128:$Rn))), + (v8bf16 (FABSv8f16 (v8bf16 V128:$Rn)))>; + def : Pat<(v4bf16 (fneg (v4bf16 V64:$Rn))), + (v4bf16 (FNEGv4f16 (v4bf16 V64:$Rn)))>; + def : Pat<(v8bf16 (fneg (v8bf16 V128:$Rn))), + (v8bf16 (FNEGv8f16 (v8bf16 V128:$Rn)))>; +} + def : Pat<(vnot (v4i16 V64:$Rn)), (NOTv8i8 V64:$Rn)>; def : Pat<(vnot (v8i16 V128:$Rn)), (NOTv16i8 V128:$Rn)>; def : Pat<(vnot (v2i32 V64:$Rn)), (NOTv8i8 V64:$Rn)>; @@ -6020,6 +6384,31 @@ defm URSQRTE: SIMDTwoVectorS<1, 1, 0b11100, "ursqrte", int_aarch64_neon_ursqrte> defm USQADD : SIMDTwoVectorBHSDTied<1, 0b00011, "usqadd",int_aarch64_neon_usqadd>; defm XTN : SIMDMixedTwoVector<0, 0b10010, "xtn", trunc>; +// Patterns for plain partial add reductions, which lower to [SU]ADALP. +multiclass SelectVectorPartialReduceAdd { + def : Pat<(DstVT (partial_reduce_smla DstVT:$Acc, SrcVT:$Input, (SrcVT immOneV))), + (!cast("SADALP" # SrcVT # "_" # DstVT) $Acc, $Input)>; + + def : Pat<(DstVT (partial_reduce_umla DstVT:$Acc, SrcVT:$Input, (SrcVT immOneV))), + (!cast("UADALP" # SrcVT # "_" # DstVT) $Acc, $Input)>; + + // A zero accumulator has nothing to add to so it lowers to [SU]ADDLP. + def : Pat<(DstVT (partial_reduce_smla (DstVT immAllZerosV), SrcVT:$Input, + (SrcVT immOneV))), + (!cast("SADDLP" # SrcVT # "_" # DstVT) $Input)>; + + def : Pat<(DstVT (partial_reduce_umla (DstVT immAllZerosV), SrcVT:$Input, + (SrcVT immOneV))), + (!cast("UADDLP" # SrcVT # "_" # DstVT) $Input)>; +} + +defm : SelectVectorPartialReduceAdd; +defm : SelectVectorPartialReduceAdd; +defm : SelectVectorPartialReduceAdd; +defm : SelectVectorPartialReduceAdd; +defm : SelectVectorPartialReduceAdd; +defm : SelectVectorPartialReduceAdd; + def : Pat<(v4f16 (AArch64rev32 V64:$Rn)), (REV32v4i16 V64:$Rn)>; def : Pat<(v4f16 (AArch64rev64 V64:$Rn)), (REV64v4i16 V64:$Rn)>; def : Pat<(v4bf16 (AArch64rev32 V64:$Rn)), (REV32v4i16 V64:$Rn)>; @@ -6659,6 +7048,8 @@ defm FCVTPU : SIMDFPTwoScalar< 1, 1, 0b11010, "fcvtpu", int_aarch64_neon_fcvtp def FCVTXNv1i64 : SIMDInexactCvtTwoScalar<0b10110, "fcvtxn">; defm FCVTZS : SIMDFPTwoScalar< 0, 1, 0b11011, "fcvtzs", any_fp_to_sint>; defm FCVTZU : SIMDFPTwoScalar< 1, 1, 0b11011, "fcvtzu", any_fp_to_uint>; +defm FCVTZS : SIMDFPScalarRShift<0, 0b11111, "fcvtzs">; +defm FCVTZU : SIMDFPScalarRShift<1, 0b11111, "fcvtzu">; defm FRECPE : SIMDFPTwoScalarNonCVT< 0, 1, 0b11101, "frecpe">; defm FRECPX : SIMDFPTwoScalarNonCVT< 0, 1, 0b11111, "frecpx">; defm FRSQRTE : SIMDFPTwoScalarNonCVT< 1, 1, 0b11101, "frsqrte">; @@ -6676,6 +7067,32 @@ defm UQXTN : SIMDTwoScalarMixedBHS<1, 0b10100, "uqxtn", int_aarch64_neon_scalar defm USQADD : SIMDTwoScalarBHSDTied< 1, 0b00011, "usqadd", AArch64usqadd, int_aarch64_neon_usqadd>; +// Scalar i32 -> i64 extend +def : Pat<(i32 (int_aarch64_neon_scalar_uqxtn (i64 FPR64:$Rn))), + (i32 (UQXTNv1i32 FPR64:$Rn))>; +def : Pat<(i32 (int_aarch64_neon_scalar_sqxtn (i64 FPR64:$Rn))), + (i32 (SQXTNv1i32 FPR64:$Rn))>; +def : Pat<(i32 (int_aarch64_neon_scalar_sqxtun (i64 FPR64:$Rn))), + (i32 (SQXTUNv1i32 FPR64:$Rn))>; + +// ssub_sat(0, R) -> sqneg(R) +def : Pat<(v16i8 (ssubsat immAllZerosV, V128:$reg)), + (v16i8 (SQNEGv16i8 V128:$reg))>; +def : Pat<(v8i16 (ssubsat immAllZerosV, V128:$reg)), + (v8i16 (SQNEGv8i16 V128:$reg))>; +def : Pat<(v4i32 (ssubsat immAllZerosV, V128:$reg)), + (v4i32 (SQNEGv4i32 V128:$reg))>; +def : Pat<(v2i64 (ssubsat immAllZerosV, V128:$reg)), + (v2i64 (SQNEGv2i64 V128:$reg))>; +def : Pat<(v8i8 (ssubsat immAllZerosV, V64:$reg)), + (v8i8 (SQNEGv8i8 V64:$reg))>; +def : Pat<(v4i16 (ssubsat immAllZerosV, V64:$reg)), + (v4i16 (SQNEGv4i16 V64:$reg))>; +def : Pat<(v2i32 (ssubsat immAllZerosV, V64:$reg)), + (v2i32 (SQNEGv2i32 V64:$reg))>; +def : Pat<(v1i64 (ssubsat immAllZerosV, V64:$reg)), + (v1i64 (SQNEGv1i64 V64:$reg))>; + // Floating-point conversion patterns. multiclass IntegerToFPSIMDScalarPatterns { let Predicates = [HasFPRCVT] in { @@ -6767,28 +7184,62 @@ multiclass FPToIntegerIntPats { def : Pat<(f64 (bitconvert (i64 (round f64:$Rn)))), (!cast(INST # v1i64) $Rn)>; } +} +defm : FPToIntegerIntPats; +defm : FPToIntegerIntPats; + +// This thing just allows us to change the type of the node produced. The imm +// value we want is already in V from SelectCVTFixedPointVec. +def fixedpoint_xform : SDNodeXForm; +def gi_fixedpoint_xform : GICustomOperandRenderer<"renderFixedPointXForm">, + GISDNodeXFormEquiv; + +def gi_fixedpoint_f16_i32 : GIComplexOperandMatcher">, + GIComplexPatternEquiv; +def gi_fixedpoint_f16_i64 : GIComplexOperandMatcher">, + GIComplexPatternEquiv; +def gi_fixedpoint_f32_i32 : GIComplexOperandMatcher">, + GIComplexPatternEquiv; +def gi_fixedpoint_f32_i64 : GIComplexOperandMatcher">, + GIComplexPatternEquiv; +def gi_fixedpoint_f64_i32 : GIComplexOperandMatcher">, + GIComplexPatternEquiv; +def gi_fixedpoint_f64_i64 : GIComplexOperandMatcher">, + GIComplexPatternEquiv; + +multiclass FPToIntegerFmulPats { let Predicates = [HasFullFP16] in { def : Pat<(i32 (round (fmul f16:$Rn, fixedpoint_f16_i32:$scale))), - (!cast(INST # SWHri) $Rn, $scale)>; + (!cast(INST # SWHri) $Rn, fixedpoint_f16_i32:$scale)>; def : Pat<(i64 (round (fmul f16:$Rn, fixedpoint_f16_i64:$scale))), - (!cast(INST # SXHri) $Rn, $scale)>; + (!cast(INST # SXHri) $Rn, fixedpoint_f16_i64:$scale)>; } def : Pat<(i32 (round (fmul f32:$Rn, fixedpoint_f32_i32:$scale))), - (!cast(INST # SWSri) $Rn, $scale)>; + (!cast(INST # SWSri) $Rn, fixedpoint_f32_i32:$scale)>; def : Pat<(i64 (round (fmul f32:$Rn, fixedpoint_f32_i64:$scale))), - (!cast(INST # SXSri) $Rn, $scale)>; + (!cast(INST # SXSri) $Rn, fixedpoint_f32_i64:$scale)>; def : Pat<(i32 (round (fmul f64:$Rn, fixedpoint_f64_i32:$scale))), - (!cast(INST # SWDri) $Rn, $scale)>; + (!cast(INST # SWDri) $Rn, fixedpoint_f64_i32:$scale)>; def : Pat<(i64 (round (fmul f64:$Rn, fixedpoint_f64_i64:$scale))), (!cast(INST # SXDri) $Rn, $scale)>; + + def : Pat<(f32 (bitconvert (i32 (round (fmul FPR32:$Rn, fixedpoint_f32_i32:$scale))))), + (!cast(INST # "s") FPR32:$Rn, (fixedpoint_xform fixedpoint_f32_i32:$scale))>; + def : Pat<(f64 (bitconvert (i64 (round (fmul FPR64:$Rn, fixedpoint_f64_i64:$scale))))), + (!cast(INST # "d") FPR64:$Rn, (fixedpoint_xform fixedpoint_f64_i64:$scale))>; } -defm : FPToIntegerIntPats; -defm : FPToIntegerIntPats; +defm : FPToIntegerFmulPats; +defm : FPToIntegerFmulPats; +defm : FPToIntegerFmulPats; +defm : FPToIntegerFmulPats; // AArch64's FCVT instructions saturate when out of range. -multiclass FPToIntegerSatPats { +multiclass FPToIntegerSatPats { let Predicates = [HasFullFP16] in { def : Pat<(i32 (to_int_sat f16:$Rn, i32)), (!cast(INST # UWHr) f16:$Rn)>; @@ -6804,38 +7255,23 @@ multiclass FPToIntegerSatPats(INST # UXDr) f64:$Rn)>; - let Predicates = [HasFullFP16] in { - def : Pat<(i32 (to_int_sat_gi f16:$Rn)), - (!cast(INST # UWHr) f16:$Rn)>; - def : Pat<(i64 (to_int_sat_gi f16:$Rn)), - (!cast(INST # UXHr) f16:$Rn)>; - } - def : Pat<(i32 (to_int_sat_gi f32:$Rn)), - (!cast(INST # UWSr) f32:$Rn)>; - def : Pat<(i64 (to_int_sat_gi f32:$Rn)), - (!cast(INST # UXSr) f32:$Rn)>; - def : Pat<(i32 (to_int_sat_gi f64:$Rn)), - (!cast(INST # UWDr) f64:$Rn)>; - def : Pat<(i64 (to_int_sat_gi f64:$Rn)), - (!cast(INST # UXDr) f64:$Rn)>; - // For global-isel we can use register classes to determine // which FCVT instruction to use. let Predicates = [HasFPRCVT] in { - def : Pat<(i32 (to_int_sat_gi f16:$Rn)), + def : Pat<(i32 (to_int_sat f16:$Rn, i32)), (!cast(INST # SHr) f16:$Rn)>; - def : Pat<(i64 (to_int_sat_gi f16:$Rn)), + def : Pat<(i64 (to_int_sat f16:$Rn, i64)), (!cast(INST # DHr) f16:$Rn)>; - def : Pat<(i64 (to_int_sat_gi f32:$Rn)), + def : Pat<(i64 (to_int_sat f32:$Rn, i64)), (!cast(INST # DSr) f32:$Rn)>; - def : Pat<(i32 (to_int_sat_gi f64:$Rn)), + def : Pat<(i32 (to_int_sat f64:$Rn, i32)), (!cast(INST # SDr) f64:$Rn)>; } let Predicates = [HasNEONandIsFPRCVTStreamingSafe] in { - def : Pat<(i32 (to_int_sat_gi f32:$Rn)), + def : Pat<(i32 (to_int_sat f32:$Rn, i32)), (!cast(INST # v1i32) f32:$Rn)>; - def : Pat<(i64 (to_int_sat_gi f64:$Rn)), + def : Pat<(i64 (to_int_sat f64:$Rn, i64)), (!cast(INST # v1i64) f64:$Rn)>; } @@ -6859,39 +7295,29 @@ multiclass FPToIntegerSatPats(INST # SWHri) $Rn, $scale)>; + (!cast(INST # SWHri) $Rn, fixedpoint_f16_i32:$scale)>; def : Pat<(i64 (to_int_sat (fmul f16:$Rn, fixedpoint_f16_i64:$scale), i64)), - (!cast(INST # SXHri) $Rn, $scale)>; + (!cast(INST # SXHri) $Rn, fixedpoint_f16_i64:$scale)>; } def : Pat<(i32 (to_int_sat (fmul f32:$Rn, fixedpoint_f32_i32:$scale), i32)), - (!cast(INST # SWSri) $Rn, $scale)>; + (!cast(INST # SWSri) $Rn, fixedpoint_f32_i32:$scale)>; def : Pat<(i64 (to_int_sat (fmul f32:$Rn, fixedpoint_f32_i64:$scale), i64)), - (!cast(INST # SXSri) $Rn, $scale)>; + (!cast(INST # SXSri) $Rn, fixedpoint_f32_i64:$scale)>; def : Pat<(i32 (to_int_sat (fmul f64:$Rn, fixedpoint_f64_i32:$scale), i32)), - (!cast(INST # SWDri) $Rn, $scale)>; + (!cast(INST # SWDri) $Rn, fixedpoint_f64_i32:$scale)>; def : Pat<(i64 (to_int_sat (fmul f64:$Rn, fixedpoint_f64_i64:$scale), i64)), - (!cast(INST # SXDri) $Rn, $scale)>; - - let Predicates = [HasFullFP16] in { - def : Pat<(i32 (to_int_sat_gi (fmul f16:$Rn, fixedpoint_f16_i32:$scale))), - (!cast(INST # SWHri) $Rn, $scale)>; - def : Pat<(i64 (to_int_sat_gi (fmul f16:$Rn, fixedpoint_f16_i64:$scale))), - (!cast(INST # SXHri) $Rn, $scale)>; - } - def : Pat<(i32 (to_int_sat_gi (fmul f32:$Rn, fixedpoint_f32_i32:$scale))), - (!cast(INST # SWSri) $Rn, $scale)>; - def : Pat<(i64 (to_int_sat_gi (fmul f32:$Rn, fixedpoint_f32_i64:$scale))), - (!cast(INST # SXSri) $Rn, $scale)>; - def : Pat<(i32 (to_int_sat_gi (fmul f64:$Rn, fixedpoint_f64_i32:$scale))), - (!cast(INST # SWDri) $Rn, $scale)>; - def : Pat<(i64 (to_int_sat_gi (fmul f64:$Rn, fixedpoint_f64_i64:$scale))), - (!cast(INST # SXDri) $Rn, $scale)>; + (!cast(INST # SXDri) $Rn, fixedpoint_f64_i64:$scale)>; + + def : Pat<(f32 (bitconvert (i32 (to_int_sat (fmul FPR32:$Rn, fixedpoint_f32_i32:$scale), i32)))), + (!cast(INST # "s") FPR32:$Rn, (fixedpoint_xform fixedpoint_f32_i32:$scale))>; + def : Pat<(f64 (bitconvert (i64 (to_int_sat (fmul FPR64:$Rn, fixedpoint_f64_i64:$scale), i64)))), + (!cast(INST # "d") FPR64:$Rn, (fixedpoint_xform fixedpoint_f64_i64:$scale))>; } -defm : FPToIntegerSatPats; -defm : FPToIntegerSatPats; +defm : FPToIntegerSatPats; +defm : FPToIntegerSatPats; -multiclass FPToIntegerPats { +multiclass FPToIntegerPats { def : Pat<(i32 (to_int (round f32:$Rn))), (!cast(INST # UWSr) f32:$Rn)>; def : Pat<(i64 (to_int (round f32:$Rn))), @@ -6944,41 +7370,26 @@ multiclass FPToIntegerPats(INST # UXDr) f64:$Rn)>; - let Predicates = [HasFullFP16] in { - def : Pat<(i32 (to_int_sat_gi (round f16:$Rn))), - (!cast(INST # UWHr) f16:$Rn)>; - def : Pat<(i64 (to_int_sat_gi (round f16:$Rn))), - (!cast(INST # UXHr) f16:$Rn)>; - } - def : Pat<(i32 (to_int_sat_gi (round f32:$Rn))), - (!cast(INST # UWSr) f32:$Rn)>; - def : Pat<(i64 (to_int_sat_gi (round f32:$Rn))), - (!cast(INST # UXSr) f32:$Rn)>; - def : Pat<(i32 (to_int_sat_gi (round f64:$Rn))), - (!cast(INST # UWDr) f64:$Rn)>; - def : Pat<(i64 (to_int_sat_gi (round f64:$Rn))), - (!cast(INST # UXDr) f64:$Rn)>; - // For global-isel we can use register classes to determine // which FCVT instruction to use. let Predicates = [HasFPRCVT] in { - def : Pat<(i32 (to_int_sat_gi (round f16:$Rn))), + def : Pat<(i32 (to_int_sat (round f16:$Rn), i32)), (!cast(INST # SHr) f16:$Rn)>; - def : Pat<(i64 (to_int_sat_gi (round f16:$Rn))), + def : Pat<(i64 (to_int_sat (round f16:$Rn), i64)), (!cast(INST # DHr) f16:$Rn)>; - def : Pat<(i64 (to_int_sat_gi (round f32:$Rn))), + def : Pat<(i64 (to_int_sat (round f32:$Rn), i64)), (!cast(INST # DSr) f32:$Rn)>; - def : Pat<(i32 (to_int_sat_gi (round f64:$Rn))), + def : Pat<(i32 (to_int_sat (round f64:$Rn), i32)), (!cast(INST # SDr) f64:$Rn)>; } let Predicates = [HasNEONandIsFPRCVTStreamingSafe] in { - def : Pat<(i32 (to_int_sat_gi (round f32:$Rn))), + def : Pat<(i32 (to_int_sat (round f32:$Rn), i32)), (!cast(INST # v1i32) f32:$Rn)>; - def : Pat<(i64 (to_int_sat_gi (round f64:$Rn))), + def : Pat<(i64 (to_int_sat (round f64:$Rn), i64)), (!cast(INST # v1i64) f64:$Rn)>; } - + let Predicates = [HasFPRCVT] in { def : Pat<(f32 (bitconvert (i32 (to_int_sat (round f16:$Rn), i32)))), (!cast(INST # SHr) f16:$Rn)>; @@ -6997,16 +7408,16 @@ multiclass FPToIntegerPats; -defm : FPToIntegerPats; -defm : FPToIntegerPats; -defm : FPToIntegerPats; -defm : FPToIntegerPats; -defm : FPToIntegerPats; -defm : FPToIntegerPats; -defm : FPToIntegerPats; -defm : FPToIntegerPats; -defm : FPToIntegerPats; +defm : FPToIntegerPats; +defm : FPToIntegerPats; +defm : FPToIntegerPats; +defm : FPToIntegerPats; +defm : FPToIntegerPats; +defm : FPToIntegerPats; +defm : FPToIntegerPats; +defm : FPToIntegerPats; +defm : FPToIntegerPats; +defm : FPToIntegerPats; // For global-isel we can use register classes to determine // which FCVT instruction to use. @@ -7099,8 +7510,8 @@ def : Pat<(f64 (bitconvert (i64 (any_llrint f64:$Rn)))), // f16 -> s16 conversions let Predicates = [HasFullFP16, HasNEONandIsFPRCVTStreamingSafe] in { -def : Pat<(i16(fp_to_sint_sat_gi f16:$Rn)), (FCVTZSv1f16 f16:$Rn)>; -def : Pat<(i16(fp_to_uint_sat_gi f16:$Rn)), (FCVTZUv1f16 f16:$Rn)>; +def : Pat<(i16(fp_to_sint_sat f16:$Rn, i16)), (FCVTZSv1f16 f16:$Rn)>; +def : Pat<(i16(fp_to_uint_sat f16:$Rn, i16)), (FCVTZUv1f16 f16:$Rn)>; } def : Pat<(v1i64 (AArch64vashr (v1i64 V64:$Rn), (i32 63))), @@ -7392,7 +7803,7 @@ def : Pat <(f64 (uint_to_fp (i32 // Advanced SIMD three different-sized vector instructions. //===----------------------------------------------------------------------===// -defm ADDHN : SIMDNarrowThreeVectorBHS<0,0b0100,"addhn", int_aarch64_neon_addhn>; +defm ADDHN : SIMDNarrowThreeVectorBHS<0,0b0100,"addhn", AArch64addhn>; defm SUBHN : SIMDNarrowThreeVectorBHS<0,0b0110,"subhn", int_aarch64_neon_subhn>; defm RADDHN : SIMDNarrowThreeVectorBHS<1,0b0100,"raddhn",int_aarch64_neon_raddhn>; defm RSUBHN : SIMDNarrowThreeVectorBHS<1,0b0110,"rsubhn",int_aarch64_neon_rsubhn>; @@ -7444,7 +7855,7 @@ multiclass Neon_mul_acc_widen_patterns { (EXTv8i8 V64:$Rn, V64:$Rm, imm:$imm)>; def : Pat<(VT128 (AArch64ext V128:$Rn, V128:$Rm, (i32 imm:$imm))), (EXTv16i8 V128:$Rn, V128:$Rm, imm:$imm)>; - // We use EXT to handle extract_subvector to copy the upper 64-bits of a - // 128-bit vector. - def : Pat<(VT64 (extract_subvector V128:$Rn, (i64 N))), - (EXTRACT_SUBREG (EXTv16i8 V128:$Rn, V128:$Rn, 8), dsub)>; // A 64-bit EXT of two halves of the same 128-bit register can be done as a // single 128-bit EXT. def : Pat<(VT64 (AArch64ext (extract_subvector V128:$Rn, (i64 0)), @@ -7969,6 +8376,15 @@ let Predicates = [HasAES] in { (i64 1), (v2i64 (PMULLv2i64 $Rn, $Rm)), (i64 0)))>; + + def : Pat<(clmulh v1i64:$Rn, v1i64:$Rm), + (EXTRACT_SUBREG (v2i64 (EXTv16i8 (v2i64 (PMULLv1i64 $Rn, $Rm)), + (v2i64 (PMULLv1i64 $Rn, $Rm)), + 8)), dsub)>; + def : Pat<(clmulh v2i64:$Rn, v2i64:$Rm), + (ZIP2v2i64 (v2i64 (PMULLv1i64 (v1i64 (EXTRACT_SUBREG $Rn, dsub)), + (v1i64 (EXTRACT_SUBREG $Rm, dsub)))), + (v2i64 (PMULLv2i64 $Rn, $Rm)))>; } def : Pat<(v16i8 (vec_ins_or_scal_vec GPR32:$Rn)), @@ -8362,18 +8778,65 @@ defm FMAXV : SIMDFPAcrossLanes<0b01111, 0, "fmaxv", AArch64fmaxv>; defm FMINNMV : SIMDFPAcrossLanes<0b01100, 1, "fminnmv", AArch64fminnmv>; defm FMINV : SIMDFPAcrossLanes<0b01111, 1, "fminv", AArch64fminv>; -def : Pat<(i32 (vector_extract (v8i16 (AArch64uaddv (v8i16 (sext (v8i8 V64:$op))))), (i64 0))), - (EXTRACT_SUBREG (v8i16 (SUBREG_TO_REG (SADDLVv8i8v V64:$op), hsub)), ssub)>; -def : Pat<(i32 (vector_extract (v8i16 (AArch64uaddv (v8i16 (zext (v8i8 V64:$op))))), (i64 0))), - (EXTRACT_SUBREG (v8i16 (SUBREG_TO_REG (UADDLVv8i8v V64:$op), hsub)), ssub)>; -def : Pat<(v8i16 (AArch64uaddv (v8i16 (sext (v8i8 V64:$op))))), - (v8i16 (SUBREG_TO_REG (SADDLVv8i8v V64:$op), hsub))>; -def : Pat<(v8i16 (AArch64uaddv (v8i16 (zext (v8i8 V64:$op))))), - (v8i16 (SUBREG_TO_REG (UADDLVv8i8v V64:$op), hsub))>; -def : Pat<(v4i32 (AArch64uaddv (v4i32 (sext (v4i16 V64:$op))))), - (v4i32 (SUBREG_TO_REG (SADDLVv4i16v V64:$op), ssub))>; -def : Pat<(v4i32 (AArch64uaddv (v4i32 (zext (v4i16 V64:$op))))), - (v4i32 (SUBREG_TO_REG (UADDLVv4i16v V64:$op), ssub))>; +multiclass SIMDAcrossLanesFPIntrinsic { + // Reductions have the unused lanes set to zero. Inserting the reduction's + // result into lane zero of a zero vector is therefore redundant. + let Predicates = [HasFullFP16] in { + def : Pat<(v4f16 (vector_insert (v4f16 immAllZerosV), + (f16 (opNode (v4f16 V64:$Rn))), (i64 0))), + (v4f16 (SUBREG_TO_REG + (!cast(acrossLanesOpc # "v4i16v") V64:$Rn), + hsub))>; + def : Pat<(v8f16 (vector_insert (v8f16 immAllZerosV), + (f16 (opNode (v8f16 V128:$Rn))), (i64 0))), + (v8f16 (SUBREG_TO_REG + (!cast(acrossLanesOpc # "v8i16v") V128:$Rn), + hsub))>; + } + def : Pat<(v4f32 (vector_insert (v4f32 immAllZerosV), + (f32 (opNode (v4f32 V128:$Rn))), (i64 0))), + (v4f32 (SUBREG_TO_REG + (!cast(acrossLanesOpc # "v4i32v") V128:$Rn), + ssub))>; +} + +defm : SIMDAcrossLanesFPIntrinsic<"FMAXV", AArch64fmaxv>; +defm : SIMDAcrossLanesFPIntrinsic<"FMINV", AArch64fminv>; +defm : SIMDAcrossLanesFPIntrinsic<"FMAXNMV", AArch64fmaxnmv>; +defm : SIMDAcrossLanesFPIntrinsic<"FMINNMV", AArch64fminnmv>; + +multiclass SIMDAcrossLaneLongReductionExt { + def : Pat<(i32 (vector_extract (v8i16 (AArch64uaddv (v8i16 (ext (v8i8 V64:$op))))), (i64 0))), + (EXTRACT_SUBREG (v8i16 (SUBREG_TO_REG (!cast(Opc#"v8i8v") V64:$op), hsub)), ssub)>; + + def : Pat<(v8i16 (AArch64uaddv (v8i16 (ext (v8i8 V64:$op))))), + (v8i16 (SUBREG_TO_REG (!cast(Opc#"v8i8v") V64:$op), hsub))>; + def : Pat<(v4i32 (AArch64uaddv (v4i32 (ext (v4i16 V64:$op))))), + (v4i32 (SUBREG_TO_REG (!cast(Opc#"v4i16v") V64:$op), ssub))>; + + // ADDLV reduction unused lanes are set to zero. Inserting into a zero vector is redundant + def : Pat<(v4i16 (vector_insert (v4i16 immAllZerosV), + (i32 (vector_extract (v8i16 (AArch64uaddv (v8i16 (ext (v8i8 V64:$Rn))))), + (i64 0))), (i64 0))), + (v4i16 (SUBREG_TO_REG (!cast(Opc#"v8i8v") V64:$Rn), hsub))>; + def : Pat<(v8i16 (vector_insert (v8i16 immAllZerosV), + (i32 (vector_extract (v8i16 (AArch64uaddv (v8i16 (ext (v8i8 V64:$Rn))))), + (i64 0))), (i64 0))), + (v8i16 (SUBREG_TO_REG (!cast(Opc#"v8i8v") V64:$Rn), hsub))>; + def : Pat<(v2i32 (vector_insert (v2i32 immAllZerosV), + (i32 (vector_extract (v4i32 (AArch64uaddv (v4i32 (ext (v4i16 V64:$Rn))))), + (i64 0))), (i64 0))), + (v2i32 (SUBREG_TO_REG (!cast(Opc#"v4i16v") V64:$Rn), ssub))>; + def : Pat<(v4i32 (vector_insert (v4i32 immAllZerosV), + (i32 (vector_extract (v4i32 (AArch64uaddv (v4i32 (ext (v4i16 V64:$Rn))))), + (i64 0))), (i64 0))), + (v4i32 (SUBREG_TO_REG (!cast(Opc#"v4i16v") V64:$Rn), ssub))>; +} + +defm : SIMDAcrossLaneLongReductionExt<"UADDLV", zext>; +defm : SIMDAcrossLaneLongReductionExt<"UADDLV", anyext>; +defm : SIMDAcrossLaneLongReductionExt<"SADDLV", sext>; multiclass SIMDAcrossLaneLongPairIntrinsic { // Patterns for addv(addlp(x)) ==> addlv @@ -8388,6 +8851,28 @@ multiclass SIMDAcrossLaneLongPairIntrinsic def : Pat<(v4i32 (AArch64uaddv (v4i32 (addlp (v8i16 V128:$op))))), (INSERT_SUBREG (v4i32 (IMPLICIT_DEF)), (!cast(Opc#"v8i16v") V128:$op), ssub)>; + // ADDLV reduction's unused lanes are set to zero. Inserting into a zero vector is redundant + def : Pat<(v4i16 (vector_insert (v4i16 immAllZerosV), + (i32 (vector_extract (v8i16 (AArch64uaddv (v8i16 (addlp (v16i8 V128:$Rn))))), + (i64 0))), (i64 0))), + (v4i16 (SUBREG_TO_REG (!cast(Opc#"v16i8v") V128:$Rn), hsub))>; + def : Pat<(v8i16 (vector_insert (v8i16 immAllZerosV), + (i32 (vector_extract (v8i16 (AArch64uaddv (v8i16 (addlp (v16i8 V128:$Rn))))), + (i64 0))), (i64 0))), + (v8i16 (SUBREG_TO_REG (!cast(Opc#"v16i8v") V128:$Rn), hsub))>; + def : Pat<(v2i32 (vector_insert (v2i32 immAllZerosV), + (i32 (vector_extract (v4i32 (AArch64uaddv (v4i32 (addlp (v8i16 V128:$Rn))))), + (i64 0))), (i64 0))), + (v2i32 (SUBREG_TO_REG (!cast(Opc#"v8i16v") V128:$Rn), ssub))>; + def : Pat<(v4i32 (vector_insert (v4i32 immAllZerosV), + (i32 (vector_extract (v4i32 (AArch64uaddv (v4i32 (addlp (v8i16 V128:$Rn))))), + (i64 0))), (i64 0))), + (v4i32 (SUBREG_TO_REG (!cast(Opc#"v8i16v") V128:$Rn), ssub))>; + def : Pat<(v2i64 (concat_vectors + (v1i64 (extract_subvector (v2i64 (AArch64uaddv (v2i64 (addlp (v4i32 V128:$Rn))))), (i64 0))), + (v1i64 immAllZerosV))), + (v2i64 (SUBREG_TO_REG (!cast(Opc#"v4i32v") V128:$Rn), dsub))>; + // Patterns for addp(addlp(x)) ==> addlv def : Pat<(v2i32 (AArch64uaddv (v2i32 (addlp (v4i16 V64:$op))))), (INSERT_SUBREG (v2i32 (IMPLICIT_DEF)), (!cast(Opc#"v4i16v") V64:$op), ssub)>; @@ -8472,7 +8957,7 @@ def : Pat<(v4i32 (opNode V128:$Rn)), // If none did, fallback to the explicit patterns, consuming the vector_extract. -def : Pat<(i32 (vector_extract (insert_subvector undef, (v8i8 (opNode V64:$Rn)), +def : Pat<(i32 (vector_extract (insert_subvector (v16i8 undef), (v8i8 (opNode V64:$Rn)), (i64 0)), (i64 0))), (EXTRACT_SUBREG (INSERT_SUBREG (v8i8 (IMPLICIT_DEF)), (!cast(!strconcat(baseOpc, "v8i8v")) V64:$Rn), @@ -8481,7 +8966,7 @@ def : Pat<(i32 (vector_extract (v16i8 (opNode V128:$Rn)), (i64 0))), (EXTRACT_SUBREG (INSERT_SUBREG (v16i8 (IMPLICIT_DEF)), (!cast(!strconcat(baseOpc, "v16i8v")) V128:$Rn), bsub), ssub)>; -def : Pat<(i32 (vector_extract (insert_subvector undef, +def : Pat<(i32 (vector_extract (insert_subvector (v8i16 undef), (v4i16 (opNode V64:$Rn)), (i64 0)), (i64 0))), (EXTRACT_SUBREG (INSERT_SUBREG (v4i16 (IMPLICIT_DEF)), (!cast(!strconcat(baseOpc, "v4i16v")) V64:$Rn), @@ -8495,14 +8980,45 @@ def : Pat<(i32 (vector_extract (v4i32 (opNode V128:$Rn)), (i64 0))), (!cast(!strconcat(baseOpc, "v4i32v")) V128:$Rn), ssub), ssub)>; -} +// Reductions have the unused lanes set to zero. Inserting the reduction's +// result into lane zero of a zero vector is therefore redundant. +def : Pat<(v16i8 (vector_insert immAllZerosV, + (i32 (vector_extract (v16i8 (opNode V128:$Rn)), (i64 0))), + (i64 0))), + (v16i8 (SUBREG_TO_REG (!cast(!strconcat(baseOpc, "v16i8v")) V128:$Rn), bsub))>; +def : Pat<(v8i16 (vector_insert immAllZerosV, + (i32 (vector_extract (v8i16 (opNode V128:$Rn)), (i64 0))), + (i64 0))), + (v8i16 (SUBREG_TO_REG (!cast(!strconcat(baseOpc, "v8i16v")) V128:$Rn), hsub))>; +def : Pat<(v4i32 (vector_insert immAllZerosV, + (i32 (vector_extract (v4i32 (opNode V128:$Rn)), (i64 0))), + (i64 0))), + (v4i32 (SUBREG_TO_REG (!cast(!strconcat(baseOpc, "v4i32v")) V128:$Rn), ssub))>; + +def : Pat<(v8i8 (vector_insert immAllZerosV, + (i32 (vector_extract (v16i8 (insert_subvector undef, + (v8i8 (opNode V64:$Rn)), (i64 0))), (i64 0))), (i64 0))), + (v8i8 (SUBREG_TO_REG (!cast(!strconcat(baseOpc, "v8i8v")) V64:$Rn), bsub))>; + +def : Pat<(v4i16 (vector_insert immAllZerosV, + (i32 (vector_extract (v8i16 (insert_subvector undef, + (v4i16 (opNode V64:$Rn)), (i64 0))), (i64 0))), (i64 0))), + (v4i16 (SUBREG_TO_REG(!cast(!strconcat(baseOpc, "v4i16v")) V64:$Rn), hsub))>; +} +// Reductions have the unused lanes set to zero. Inserting the reduction's +// result into lane zero of a zero vector is therefore redundant. +def : Pat<(v2i64 (concat_vectors (v1i64 (extract_subvector (v2i64 ( AArch64uaddv (v2i64 V128:$Rn))), (i64 0))), + immAllZerosV)), + (v2i64 (SUBREG_TO_REG (ADDPv2i64p V128:$Rn), dsub))>; + + multiclass SIMDAcrossLanesSignedIntrinsic : SIMDAcrossLanesIntrinsic { // If there is a sign extension after this intrinsic, consume it as smov already // performed it -def : Pat<(i32 (sext_inreg (i32 (vector_extract (insert_subvector undef, +def : Pat<(i32 (sext_inreg (i32 (vector_extract (insert_subvector (v16i8 undef), (opNode (v8i8 V64:$Rn)), (i64 0)), (i64 0))), i8)), (i32 (SMOVvi8to32 (INSERT_SUBREG (v16i8 (IMPLICIT_DEF)), @@ -8514,7 +9030,7 @@ def : Pat<(i32 (sext_inreg (i32 (vector_extract (INSERT_SUBREG (v16i8 (IMPLICIT_DEF)), (!cast(!strconcat(baseOpc, "v16i8v")) V128:$Rn), bsub), (i64 0)))>; -def : Pat<(i32 (sext_inreg (i32 (vector_extract (insert_subvector undef, +def : Pat<(i32 (sext_inreg (i32 (vector_extract (insert_subvector (v8i16 undef), (opNode (v4i16 V64:$Rn)), (i64 0)), (i64 0))), i16)), (i32 (SMOVvi16to32 (INSERT_SUBREG (v16i8 (IMPLICIT_DEF)), @@ -8533,7 +9049,7 @@ multiclass SIMDAcrossLanesUnsignedIntrinsic { // If there is a masking operation keeping only what has been actually // generated, consume it. -def : Pat<(i32 (and (i32 (vector_extract (insert_subvector undef, +def : Pat<(i32 (and (i32 (vector_extract (insert_subvector (v16i8 undef), (opNode (v8i8 V64:$Rn)), (i64 0)), (i64 0))), maski8_or_more)), (i32 (EXTRACT_SUBREG (INSERT_SUBREG (v16i8 (IMPLICIT_DEF)), @@ -8545,7 +9061,7 @@ def : Pat<(i32 (and (i32 (vector_extract (opNode (v16i8 V128:$Rn)), (i64 0))), (INSERT_SUBREG (v16i8 (IMPLICIT_DEF)), (!cast(!strconcat(baseOpc, "v16i8v")) V128:$Rn), bsub), ssub))>; -def : Pat<(i32 (and (i32 (vector_extract (insert_subvector undef, +def : Pat<(i32 (and (i32 (vector_extract (insert_subvector (v8i16 undef), (opNode (v4i16 V64:$Rn)), (i64 0)), (i64 0))), maski16_or_more)), (i32 (EXTRACT_SUBREG (INSERT_SUBREG (v16i8 (IMPLICIT_DEF)), @@ -8645,6 +9161,7 @@ def : Pat<(v2i64 (AArch64saddlv (v2i32 V64:$Rn))), def : Pat<(v2i64 (AArch64uaddlv (v2i32 V64:$Rn))), (v2i64 (INSERT_SUBREG (v2i64 (IMPLICIT_DEF)), (UADDLPv2i32_v1i64 V64:$Rn), dsub))>; +// uaddlp from shuffle + zext def : Pat<(add (AArch64bici (v8i16 (bitconvert v16i8:$src)), (i32 255), (i32 8)), (AArch64vlshr (v8i16 (AArch64NvCast v16i8:$src)), (i32 8))), (UADDLPv16i8_v8i16 V128:$src)>; @@ -8660,6 +9177,44 @@ def : Pat<(add (v2i64 (zext (v2i32 (AArch64zip1 (v2i32 (extract_subvector v4i32: (v2i32 (extract_subvector v4i32:$src, (i64 2)))))))), (UADDLPv4i32_v2i64 V128:$src)>; +// uaddlp from zext + shuffle +def : Pat<(add (AArch64uzp1 (v8i16 (zext (v8i8 (extract_subvector v16i8:$src, (i64 0))))), + (v8i16 (zext (v8i8 (extract_subvector v16i8:$src, (i64 8)))))), + (AArch64uzp2 (v8i16 (zext (v8i8 (extract_subvector v16i8:$src, (i64 0))))), + (v8i16 (zext (v8i8 (extract_subvector v16i8:$src, (i64 8))))))), + (UADDLPv16i8_v8i16 V128:$src)>; + +def : Pat<(add (AArch64uzp1 (v4i32 (zext (v4i16 (extract_subvector v8i16:$src, (i64 0))))), + (v4i32 (zext (v4i16 (extract_subvector v8i16:$src, (i64 4)))))), + (AArch64uzp2 (v4i32 (zext (v4i16 (extract_subvector v8i16:$src, (i64 0))))), + (v4i32 (zext (v4i16 (extract_subvector v8i16:$src, (i64 4))))))), + (UADDLPv8i16_v4i32 V128:$src)>; + +def : Pat<(add (AArch64zip1 (v2i64 (zext (v2i32 (extract_subvector v4i32:$src, (i64 0))))), + (v2i64 (zext (v2i32 (extract_subvector v4i32:$src, (i64 2)))))), + (AArch64zip2 (v2i64 (zext (v2i32 (extract_subvector v4i32:$src, (i64 0))))), + (v2i64 (zext (v2i32 (extract_subvector v4i32:$src, (i64 2))))))), + (UADDLPv4i32_v2i64 V128:$src)>; + +// saddlp from sext + shuffle +def : Pat<(add (AArch64uzp1 (v8i16 (sext (v8i8 (extract_subvector v16i8:$src, (i64 0))))), + (v8i16 (sext (v8i8 (extract_subvector v16i8:$src, (i64 8)))))), + (AArch64uzp2 (v8i16 (sext (v8i8 (extract_subvector v16i8:$src, (i64 0))))), + (v8i16 (sext (v8i8 (extract_subvector v16i8:$src, (i64 8))))))), + (SADDLPv16i8_v8i16 V128:$src)>; + +def : Pat<(add (AArch64uzp1 (v4i32 (sext (v4i16 (extract_subvector v8i16:$src, (i64 0))))), + (v4i32 (sext (v4i16 (extract_subvector v8i16:$src, (i64 4)))))), + (AArch64uzp2 (v4i32 (sext (v4i16 (extract_subvector v8i16:$src, (i64 0))))), + (v4i32 (sext (v4i16 (extract_subvector v8i16:$src, (i64 4))))))), + (SADDLPv8i16_v4i32 V128:$src)>; + +def : Pat<(add (AArch64zip1 (v2i64 (sext (v2i32 (extract_subvector v4i32:$src, (i64 0))))), + (v2i64 (sext (v2i32 (extract_subvector v4i32:$src, (i64 2)))))), + (AArch64zip2 (v2i64 (sext (v2i32 (extract_subvector v4i32:$src, (i64 0))))), + (v2i64 (sext (v2i32 (extract_subvector v4i32:$src, (i64 2))))))), + (SADDLPv4i32_v2i64 V128:$src)>; + //------------------------------------------------------------------------------ // AdvSIMD modified immediate instructions //------------------------------------------------------------------------------ @@ -9005,6 +9560,88 @@ defm : FMLSIndexedAfterNegPatterns< defm : FMLSIndexedAfterNegPatterns< TriOpFrag<(any_fma node:$MHS, node:$RHS, node:$LHS)> >; +// Vector FNMUL: -(x * y) -> FMLS(-0.0, x, y), by-element for a splat y. +// AArch64 has no vector FNMUL, so -(x * y) needs an FNEG on top of the FMUL. +// FMLS folds it in, and the MOVI has no input dependency, so the FNEG leaves +// the dependency chain. The consumed FNEG/FMUL has to be single-use, or it +// stays live and the MOVI is added rather than substituted. +// Like the scalar FNMUL patterns above this changes the sign of a NaN result +// (FMLS negates its multiplicand); LangRef leaves that sign non-deterministic. +// Plain fmul only: the two forms agree only when negation commutes with +// rounding, which fails for the directed modes constrained FP can run under. +// Only f32 and f16 are covered; f64 -0.0 is not a MOVI immediate. +let HasOneUse = 1 in { + def fneg_oneuse : PatFrag<(ops node:$src), (fneg node:$src)>; + def fmul_oneuse : PatFrag<(ops node:$lhs, node:$rhs), + (fmul node:$lhs, node:$rhs)>; +} + +// -(Rn * Rm) in all its forms. Only the by-element patterns below look at +// splats; they take Rm apart for the lane operand, so Rn is bound to the FNEG'd +// operand -- the one a split FNEG+FMUL materializes anyway. Both FMUL orders +// are listed because TableGen's commuted copies sort after these. +def fnmul_frags : PatFrags<(ops node:$Rn, node:$Rm), + [(fneg (fmul_oneuse node:$Rn, node:$Rm)), + (fmul (fneg_oneuse node:$Rn), node:$Rm), + (fmul node:$Rm, (fneg_oneuse node:$Rn)), + // Lane slot for the FNEG'd operand, when it is the + // only splat. + (fmul node:$Rn, (fneg_oneuse node:$Rm))]>; + +let Predicates = [HasNEON, HasFullFP16] in { + def : Pat<(v8f16 (fnmul_frags (v8f16 V128:$Rn), + (v8f16 (AArch64duplane16 (v8f16 V128_lo:$Rm), + VectorIndexH:$idx)))), + (FMLSv8i16_indexed (MOVIv8i16 (i32 128), (i32 8)), V128:$Rn, + V128_lo:$Rm, VectorIndexH:$idx)>; + def : Pat<(v8f16 (fnmul_frags (v8f16 V128:$Rn), + (v8f16 (AArch64dup (f16 FPR16Op_lo:$Rm))))), + (FMLSv8i16_indexed (MOVIv8i16 (i32 128), (i32 8)), V128:$Rn, + (SUBREG_TO_REG (f16 FPR16Op_lo:$Rm), hsub), + (i64 0))>; + def : Pat<(v4f16 (fnmul_frags (v4f16 V64:$Rn), + (v4f16 (AArch64duplane16 (v8f16 V128_lo:$Rm), + VectorIndexH:$idx)))), + (FMLSv4i16_indexed (MOVIv4i16 (i32 128), (i32 8)), V64:$Rn, + V128_lo:$Rm, VectorIndexH:$idx)>; + def : Pat<(v4f16 (fnmul_frags (v4f16 V64:$Rn), + (v4f16 (AArch64dup (f16 FPR16Op_lo:$Rm))))), + (FMLSv4i16_indexed (MOVIv4i16 (i32 128), (i32 8)), V64:$Rn, + (SUBREG_TO_REG (f16 FPR16Op_lo:$Rm), hsub), + (i64 0))>; + + def : Pat<(v8f16 (fnmul_frags (v8f16 V128:$Rn), (v8f16 V128:$Rm))), + (FMLSv8f16 (MOVIv8i16 (i32 128), (i32 8)), V128:$Rn, V128:$Rm)>; + def : Pat<(v4f16 (fnmul_frags (v4f16 V64:$Rn), (v4f16 V64:$Rm))), + (FMLSv4f16 (MOVIv4i16 (i32 128), (i32 8)), V64:$Rn, V64:$Rm)>; +} + +let Predicates = [HasNEON] in { + def : Pat<(v4f32 (fnmul_frags (v4f32 V128:$Rn), + (v4f32 (AArch64duplane32 (v4f32 V128:$Rm), + VectorIndexS:$idx)))), + (FMLSv4i32_indexed (MOVIv4i32 (i32 128), (i32 24)), V128:$Rn, + V128:$Rm, VectorIndexS:$idx)>; + def : Pat<(v4f32 (fnmul_frags (v4f32 V128:$Rn), + (v4f32 (AArch64dup (f32 FPR32Op:$Rm))))), + (FMLSv4i32_indexed (MOVIv4i32 (i32 128), (i32 24)), V128:$Rn, + (SUBREG_TO_REG FPR32Op:$Rm, ssub), (i64 0))>; + def : Pat<(v2f32 (fnmul_frags (v2f32 V64:$Rn), + (v2f32 (AArch64duplane32 (v4f32 V128:$Rm), + VectorIndexS:$idx)))), + (FMLSv2i32_indexed (MOVIv2i32 (i32 128), (i32 24)), V64:$Rn, + V128:$Rm, VectorIndexS:$idx)>; + def : Pat<(v2f32 (fnmul_frags (v2f32 V64:$Rn), + (v2f32 (AArch64dup (f32 FPR32Op:$Rm))))), + (FMLSv2i32_indexed (MOVIv2i32 (i32 128), (i32 24)), V64:$Rn, + (SUBREG_TO_REG FPR32Op:$Rm, ssub), (i64 0))>; + + def : Pat<(v4f32 (fnmul_frags (v4f32 V128:$Rn), (v4f32 V128:$Rm))), + (FMLSv4f32 (MOVIv4i32 (i32 128), (i32 24)), V128:$Rn, V128:$Rm)>; + def : Pat<(v2f32 (fnmul_frags (v2f32 V64:$Rn), (v2f32 V64:$Rm))), + (FMLSv2f32 (MOVIv2i32 (i32 128), (i32 24)), V64:$Rn, V64:$Rm)>; +} + defm FMULX : SIMDFPIndexed<1, 0b1001, "fmulx", int_aarch64_neon_fmulx>; defm FMUL : SIMDFPIndexed<0, 0b1001, "fmul", any_fmul>; @@ -9076,27 +9713,50 @@ def : Pat<(i64 (int_aarch64_neon_sqsub (i64 FPR64:$Rd), //---------------------------------------------------------------------------- // AdvSIMD scalar shift instructions //---------------------------------------------------------------------------- -defm FCVTZS : SIMDFPScalarRShift<0, 0b11111, "fcvtzs">; -defm FCVTZU : SIMDFPScalarRShift<1, 0b11111, "fcvtzu">; defm SCVTF : SIMDFPScalarRShift<0, 0b11100, "scvtf">; defm UCVTF : SIMDFPScalarRShift<1, 0b11100, "ucvtf">; -// Codegen patterns for the above. We don't put these directly on the + +// Transformation for SIMD shift imm to fixed point imm for FPR-to-GPR result. +def fixedpoint_scalar_xform : SDNodeXForm; +def gi_fixedpoint_scalar_xform + : GICustomOperandRenderer<"renderFixedPointScalarXForm">, + GISDNodeXFormEquiv; + +multiclass FPToFixedScalarPats { + // Allow integer result to remain in GPR register. + def : Pat<(i32 (OpN FPR32:$Rn, vecshiftR32:$imm)), + (!cast(INST # "SWSri") FPR32:$Rn, (fixedpoint_scalar_xform vecshiftR32:$imm))>; + def : Pat<(i64 (OpN (f64 FPR64:$Rn), vecshiftR64:$imm)), + (!cast(INST # "SXDri") FPR64:$Rn, (fixedpoint_scalar_xform vecshiftR64:$imm))>; + + // Explicit Bitcast results kept in FP/SIMD registers. + def : Pat<(f32 (bitconvert(i32 (OpN FPR32:$Rn, vecshiftR32:$imm)))), + (!cast(INST # "s") FPR32:$Rn, vecshiftR32:$imm)>; + def : Pat<(f64 (bitconvert(i64 (OpN (f64 FPR64:$Rn), vecshiftR64:$imm)))), + (!cast(INST # "d") FPR64:$Rn, vecshiftR64:$imm)>; + + def : Pat<(v1i64 (OpN (v1f64 FPR64:$Rn), vecshiftR64:$imm)), + (!cast(INST # "d") FPR64:$Rn, vecshiftR64:$imm)>; + def : Pat<(i32 (OpN (f16 FPR16:$Rn), vecshiftR32:$imm)), + (i32 (INSERT_SUBREG + (i32 (IMPLICIT_DEF)), + (!cast(INST # "h") FPR16:$Rn, vecshiftR32:$imm), + hsub))>; + def : Pat<(i64 (OpN (f16 FPR16:$Rn), vecshiftR64:$imm)), + (i64 (INSERT_SUBREG + (i64 (IMPLICIT_DEF)), + (!cast(INST # "h") FPR16:$Rn, vecshiftR64:$imm), + hsub))>; +} +defm : FPToFixedScalarPats; +defm : FPToFixedScalarPats; + +// Codegen patterns for SCVTF and UCVTF. We don't put these directly on the // instructions because TableGen's type inference can't handle the truth. // Having the same base pattern for fp <--> int totally freaks it out. -def : Pat<(int_aarch64_neon_vcvtfp2fxs FPR32:$Rn, vecshiftR32:$imm), - (FCVTZSs FPR32:$Rn, vecshiftR32:$imm)>; -def : Pat<(int_aarch64_neon_vcvtfp2fxu FPR32:$Rn, vecshiftR32:$imm), - (FCVTZUs FPR32:$Rn, vecshiftR32:$imm)>; -def : Pat<(i64 (int_aarch64_neon_vcvtfp2fxs (f64 FPR64:$Rn), vecshiftR64:$imm)), - (FCVTZSd FPR64:$Rn, vecshiftR64:$imm)>; -def : Pat<(i64 (int_aarch64_neon_vcvtfp2fxu (f64 FPR64:$Rn), vecshiftR64:$imm)), - (FCVTZUd FPR64:$Rn, vecshiftR64:$imm)>; -def : Pat<(v1i64 (int_aarch64_neon_vcvtfp2fxs (v1f64 FPR64:$Rn), - vecshiftR64:$imm)), - (FCVTZSd FPR64:$Rn, vecshiftR64:$imm)>; -def : Pat<(v1i64 (int_aarch64_neon_vcvtfp2fxu (v1f64 FPR64:$Rn), - vecshiftR64:$imm)), - (FCVTZUd FPR64:$Rn, vecshiftR64:$imm)>; def : Pat<(int_aarch64_neon_vcvtfxu2fp FPR32:$Rn, vecshiftR32:$imm), (UCVTFs FPR32:$Rn, vecshiftR32:$imm)>; def : Pat<(f64 (int_aarch64_neon_vcvtfxu2fp (i64 FPR64:$Rn), vecshiftR64:$imm)), @@ -9128,26 +9788,6 @@ def : Pat<(f16 (int_aarch64_neon_vcvtfxu2fp FPR32:$Rn, vecshiftR16:$imm)), (UCVTFh (f16 (EXTRACT_SUBREG FPR32:$Rn, hsub)), vecshiftR16:$imm)>; def : Pat<(f16 (int_aarch64_neon_vcvtfxu2fp (i64 FPR64:$Rn), vecshiftR16:$imm)), (UCVTFh (f16 (EXTRACT_SUBREG FPR64:$Rn, hsub)), vecshiftR16:$imm)>; -def : Pat<(i32 (int_aarch64_neon_vcvtfp2fxs (f16 FPR16:$Rn), vecshiftR32:$imm)), - (i32 (INSERT_SUBREG - (i32 (IMPLICIT_DEF)), - (FCVTZSh FPR16:$Rn, vecshiftR32:$imm), - hsub))>; -def : Pat<(i64 (int_aarch64_neon_vcvtfp2fxs (f16 FPR16:$Rn), vecshiftR64:$imm)), - (i64 (INSERT_SUBREG - (i64 (IMPLICIT_DEF)), - (FCVTZSh FPR16:$Rn, vecshiftR64:$imm), - hsub))>; -def : Pat<(i32 (int_aarch64_neon_vcvtfp2fxu (f16 FPR16:$Rn), vecshiftR32:$imm)), - (i32 (INSERT_SUBREG - (i32 (IMPLICIT_DEF)), - (FCVTZUh FPR16:$Rn, vecshiftR32:$imm), - hsub))>; -def : Pat<(i64 (int_aarch64_neon_vcvtfp2fxu (f16 FPR16:$Rn), vecshiftR64:$imm)), - (i64 (INSERT_SUBREG - (i64 (IMPLICIT_DEF)), - (FCVTZUh FPR16:$Rn, vecshiftR64:$imm), - hsub))>; def : Pat<(i32 (int_aarch64_neon_facge (f16 FPR16:$Rn), (f16 FPR16:$Rm))), (i32 (INSERT_SUBREG (i32 (IMPLICIT_DEF)), @@ -9210,14 +9850,6 @@ class fixedpoint_vec_f32 : ComplexPattern">; class fixedpoint_vec_f16 : ComplexPattern">; -// This thing just allows us to change the type of the node produced. The imm -// value we want is already in V from SelectCVTFixedPointVec. -def fixedpoint_vec_xform : SDNodeXForm; -def gi_fixedpoint_vec_xform : GICustomOperandRenderer<"renderFixedPointXForm">, - GISDNodeXFormEquiv; def fixedpoint_v2f64 : fixedpoint_vec_f64; def fixedpoint_v2f32 : fixedpoint_vec_f32; @@ -9258,32 +9890,24 @@ multiclass FCVTPat("FCVTZS"#IVT#"_shift") (FVT RC:$Vn), (i32 1))>; def : Pat<(IVT (fp_to_sint_sat (FVT (fadd RC:$Vn, RC:$Vn)), ScalarVT)), (!cast("FCVTZS"#IVT#"_shift") (FVT RC:$Vn), (i32 1))>; - def : Pat<(IVT (fp_to_sint_sat_gi (FVT (fadd RC:$Vn, RC:$Vn)))), - (!cast("FCVTZS"#IVT#"_shift") (FVT RC:$Vn), (i32 1))>; // fptoui(fadd(x, x)) -> FCVTZU_shift x, 1 def : Pat<(IVT (fp_to_uint (FVT (fadd RC:$Vn, RC:$Vn)))), (!cast("FCVTZU"#IVT#"_shift") (FVT RC:$Vn), (i32 1))>; def : Pat<(IVT (fp_to_uint_sat (FVT (fadd RC:$Vn, RC:$Vn)), ScalarVT)), (!cast("FCVTZU"#IVT#"_shift") (FVT RC:$Vn), (i32 1))>; - def : Pat<(IVT (fp_to_uint_sat_gi (FVT (fadd RC:$Vn, RC:$Vn)))), - (!cast("FCVTZU"#IVT#"_shift") (FVT RC:$Vn), (i32 1))>; // fptosi(fmul(x, power2)) -> FCVTZS_shift x, log2(power2) def : Pat<(IVT (fp_to_sint (FVT (fmul FVT:$Vn, fixedpoint:$scale)))), - (!cast("FCVTZS"#IVT#"_shift") $Vn, (fixedpoint_vec_xform fixedpoint:$scale))>; + (!cast("FCVTZS"#IVT#"_shift") $Vn, (fixedpoint_xform fixedpoint:$scale))>; def : Pat<(IVT (fp_to_sint_sat (FVT (fmul FVT:$Vn, fixedpoint:$scale)), ScalarVT)), - (!cast("FCVTZS"#IVT#"_shift") $Vn, (fixedpoint_vec_xform fixedpoint:$scale))>; - def : Pat<(IVT (fp_to_sint_sat_gi (FVT (fmul FVT:$Vn, fixedpoint:$scale)))), - (!cast("FCVTZS"#IVT#"_shift") $Vn, (fixedpoint_vec_xform fixedpoint:$scale))>; + (!cast("FCVTZS"#IVT#"_shift") $Vn, (fixedpoint_xform fixedpoint:$scale))>; // fptoui(fmul(x, power2)) -> FCVTZU_shift x, log2(power2) def : Pat<(IVT (fp_to_uint (FVT (fmul FVT:$Vn, fixedpoint:$scale)))), - (!cast("FCVTZU"#IVT#"_shift") $Vn, (fixedpoint_vec_xform fixedpoint:$scale))>; + (!cast("FCVTZU"#IVT#"_shift") $Vn, (fixedpoint_xform fixedpoint:$scale))>; def : Pat<(IVT (fp_to_uint_sat (FVT (fmul FVT:$Vn, fixedpoint:$scale)), ScalarVT)), - (!cast("FCVTZU"#IVT#"_shift") $Vn, (fixedpoint_vec_xform fixedpoint:$scale))>; - def : Pat<(IVT (fp_to_uint_sat_gi (FVT (fmul FVT:$Vn, fixedpoint:$scale)))), - (!cast("FCVTZU"#IVT#"_shift") $Vn, (fixedpoint_vec_xform fixedpoint:$scale))>; + (!cast("FCVTZU"#IVT#"_shift") $Vn, (fixedpoint_xform fixedpoint:$scale))>; } let Predicates = [HasNEON] in { @@ -9491,19 +10115,42 @@ def : Pat<(v4i32 (concat_vectors (v2i32 V64:$Rd), (SHRNv4i32_shift (INSERT_SUBREG (IMPLICIT_DEF), V64:$Rd, dsub), V128:$Rn, vecshiftR32Narrow:$imm)>; -def : Pat<(shl (v8i16 (zext (v8i8 V64:$Rm))), (v8i16 (AArch64dup (i32 imm32_0_7:$size)))), - (USHLLv8i8_shift V64:$Rm, (i32 imm32_0_7:$size))>; -def : Pat<(shl (v4i32 (zext (v4i16 V64:$Rm))), (v4i32 (AArch64dup (i32 imm32_0_15:$size)))), - (USHLLv4i16_shift V64:$Rm, (i32 imm32_0_15:$size))>; -def : Pat<(shl (v2i64 (zext (v2i32 V64:$Rm))), (v2i64 (AArch64dup (i64 imm0_31:$size)))), - (USHLLv2i32_shift V64:$Rm, (trunc_imm imm0_31:$size))>; - -def : Pat<(shl (v8i16 (sext (v8i8 V64:$Rm))), (v8i16 (AArch64dup (i32 imm32_0_7:$size)))), - (SSHLLv8i8_shift V64:$Rm, (i32 imm32_0_7:$size))>; -def : Pat<(shl (v4i32 (sext (v4i16 V64:$Rm))), (v4i32 (AArch64dup (i32 imm32_0_15:$size)))), - (SSHLLv4i16_shift V64:$Rm, (i32 imm32_0_15:$size))>; -def : Pat<(shl (v2i64 (sext (v2i32 V64:$Rm))), (v2i64 (AArch64dup (i64 imm0_31:$size)))), - (SSHLLv2i32_shift V64:$Rm, (trunc_imm imm0_31:$size))>; +multiclass WidenShiftImmPats { + def : Pat<(shl (v8i16 (ext (v8i8 V64:$Rm))), (v8i16 (AArch64dup (i32 imm32_0_7:$size)))), + (!cast(inst#"v8i8_shift") V64:$Rm, (i32 imm32_0_7:$size))>; + def : Pat<(shl (v4i32 (ext (v4i16 V64:$Rm))), (v4i32 (AArch64dup (i32 imm32_0_15:$size)))), + (!cast(inst#"v4i16_shift") V64:$Rm, (i32 imm32_0_15:$size))>; + def : Pat<(shl (v2i64 (ext (v2i32 V64:$Rm))), (v2i64 (AArch64dup (i64 imm0_31:$size)))), + (!cast(inst#"v2i32_shift") V64:$Rm, (trunc_imm imm0_31:$size))>; + + def : Pat<(shl (v8i16 (ext (extract_high_v16i8 (v16i8 V128:$Rn)))), (v8i16 (AArch64dup (i32 imm32_0_7:$size)))), + (!cast(inst#"v16i8_shift") V128:$Rn, (i32 imm32_0_7:$size))>; + def : Pat<(shl (v4i32 (ext (extract_high_v8i16 (v8i16 V128:$Rn)))), (v4i32 (AArch64dup (i32 imm32_0_15:$size)))), + (!cast(inst#"v8i16_shift") V128:$Rn, (i32 imm32_0_15:$size))>; + def : Pat<(shl (v2i64 (ext (extract_high_v4i32 (v4i32 V128:$Rn)))), (v2i64 (AArch64dup (i64 imm0_31:$size)))), + (!cast(inst#"v4i32_shift") V128:$Rn, (trunc_imm imm0_31:$size))>; +} +defm : WidenShiftImmPats<"SSHLL", sext>; +defm : WidenShiftImmPats<"USHLL", zext>; +defm : WidenShiftImmPats<"USHLL", anyext>; + +multiclass WidenShiftFullWidthPats { + def : Pat<(shl (v8i16 (ext (v8i8 V64:$Rm))), (v8i16 (AArch64dup (i32 8)))), + (SHLLv8i8 V64:$Rm)>; + def : Pat<(shl (v4i32 (ext (v4i16 V64:$Rm))), (v4i32 (AArch64dup (i32 16)))), + (SHLLv4i16 V64:$Rm)>; + def : Pat<(shl (v2i64 (ext (v2i32 V64:$Rm))), (v2i64 (AArch64dup (i64 32)))), + (SHLLv2i32 V64:$Rm)>; + def : Pat<(shl (v8i16 (ext (extract_high_v16i8 (v16i8 V128:$Rn)))), (v8i16 (AArch64dup (i32 8)))), + (SHLLv16i8 V128:$Rn)>; + def : Pat<(shl (v4i32 (ext (extract_high_v8i16 (v8i16 V128:$Rn)))), (v4i32 (AArch64dup (i32 16)))), + (SHLLv8i16 V128:$Rn)>; + def : Pat<(shl (v2i64 (ext (extract_high_v4i32 (v4i32 V128:$Rn)))), (v2i64 (AArch64dup (i64 32)))), + (SHLLv4i32 V128:$Rn)>; +} +defm : WidenShiftFullWidthPats; +defm : WidenShiftFullWidthPats; +defm : WidenShiftFullWidthPats; // Vector sign and zero extensions are implemented with SSHLL and USSHLL. // Anyexts are implemented as zexts. @@ -9809,7 +10456,7 @@ class St1PostPat : Pat<(post_store ty:$Vt, GPR64sp:$Rn, (i64 off)), (INST ty:$Vt, GPR64sp:$Rn, XZR)>; -let Predicates = [IsBE] in { +let Predicates = [IsBEOrDisallowMisalignedMemAccesses] in { def : St1PostPat; def : St1PostPat; def : St1PostPat; @@ -10098,6 +10745,13 @@ def : St1Lane128SubvecPat; def : St1Lane128SubvecPat; def : St1Lane128SubvecPat; +// Truncating post-inc stores from lane 0 of v4i32/v2i64. +defm : St1LanePost128Pat; +defm : St1LanePost128Pat; +defm : St1LanePost128Pat; +defm : St1LanePost128Pat; +defm : St1LanePost128Pat; + let mayStore = 1, hasSideEffects = 0 in { defm ST2 : SIMDStSingleB<1, 0b000, "st2", VecListTwob, GPR64pi2>; defm ST2 : SIMDStSingleH<1, 0b010, 0, "st2", VecListTwoh, GPR64pi4>; @@ -10308,6 +10962,27 @@ def : Pat<(v8i16 (AArch64sqdmulh (v8i16 V128:$Rn), (v8i16 V128:$Rm))), def : Pat<(v4i32 (AArch64sqdmulh (v4i32 V128:$Rn), (v4i32 V128:$Rm))), (SQDMULHv4i32 V128:$Rn, V128:$Rm)>; +def AArch64sqdmulhPat : PatFrag<(ops node:$Rn, node:$Rm, node:$offset), + (truncssat_s (AArch64vashr (AArch64smull node:$Rn, node:$Rm), node:$offset))>; +def : Pat<(v4i16 (AArch64sqdmulhPat (v4i16 V64:$Rn), (v4i16 V64:$Rm), (i32 15))), + (SQDMULHv4i16 V64:$Rn, V64:$Rm)>; +def : Pat<(v2i32 (AArch64sqdmulhPat (v2i32 V64:$Rn), (v2i32 V64:$Rm), (i32 31))), + (SQDMULHv2i32 V64:$Rn, V64:$Rm)>; +def : Pat<(v8i16 (concat_vectors (v4i16 (AArch64sqdmulhPat (v4i16 (extract_subvector V128:$Rn, (i64 0))), + (v4i16 (extract_subvector V128:$Rm, (i64 0))), + (i32 15))), + (v4i16 (AArch64sqdmulhPat (v4i16 (extract_subvector V128:$Rn, (i64 4))), + (v4i16 (extract_subvector V128:$Rm, (i64 4))), + (i32 15))))), + (SQDMULHv8i16 V128:$Rn, V128:$Rm)>; +def : Pat<(v4i32 (concat_vectors (v2i32 (AArch64sqdmulhPat (v2i32 (extract_subvector V128:$Rn, (i64 0))), + (v2i32 (extract_subvector V128:$Rm, (i64 0))), + (i32 31))), + (v2i32 (AArch64sqdmulhPat (v2i32 (extract_subvector V128:$Rn, (i64 2))), + (v2i32 (extract_subvector V128:$Rm, (i64 2))), + (i32 31))))), + (SQDMULHv4i32 V128:$Rn, V128:$Rm)>; + // Conversions within AdvSIMD types in the same register size are free. // But because we need a consistent lane ordering, in big endian many // conversions require one or more REV instructions. @@ -10951,14 +11626,22 @@ def : Pat<(v1i64 (extract_subvector V128:$Rn, (i64 0))), def : Pat<(v1f64 (extract_subvector V128:$Rn, (i64 0))), (EXTRACT_SUBREG V128:$Rn, dsub)>; -def : Pat<(v8i8 (extract_subvector (v16i8 FPR128:$Rn), (i64 1))), - (EXTRACT_SUBREG (DUPv2i64lane FPR128:$Rn, 1), dsub)>; -def : Pat<(v4i16 (extract_subvector (v8i16 FPR128:$Rn), (i64 1))), - (EXTRACT_SUBREG (DUPv2i64lane FPR128:$Rn, 1), dsub)>; -def : Pat<(v2i32 (extract_subvector (v4i32 FPR128:$Rn), (i64 1))), - (EXTRACT_SUBREG (DUPv2i64lane FPR128:$Rn, 1), dsub)>; +def : Pat<(v8i8 (extract_subvector (v16i8 FPR128:$Rn), (i64 8))), + (DUPi64 FPR128:$Rn, 1)>; +def : Pat<(v4i16 (extract_subvector (v8i16 FPR128:$Rn), (i64 4))), + (DUPi64 FPR128:$Rn, 1)>; +def : Pat<(v2i32 (extract_subvector (v4i32 FPR128:$Rn), (i64 2))), + (DUPi64 FPR128:$Rn, 1)>; def : Pat<(v1i64 (extract_subvector (v2i64 FPR128:$Rn), (i64 1))), - (EXTRACT_SUBREG (DUPv2i64lane FPR128:$Rn, 1), dsub)>; + (DUPi64 FPR128:$Rn, 1)>; +def : Pat<(v4bf16 (extract_subvector (v8bf16 FPR128:$Rn), (i64 4))), + (DUPi64 FPR128:$Rn, 1)>; +def : Pat<(v4f16 (extract_subvector (v8f16 FPR128:$Rn), (i64 4))), + (DUPi64 FPR128:$Rn, 1)>; +def : Pat<(v2f32 (extract_subvector (v4f32 FPR128:$Rn), (i64 2))), + (DUPi64 FPR128:$Rn, 1)>; +def : Pat<(v1f64 (extract_subvector (v2f64 FPR128:$Rn), (i64 1))), + (DUPi64 FPR128:$Rn, 1)>; // A 64-bit subvector insert to the first 128-bit vector position // is a subregister copy that needs no instruction. @@ -11363,12 +12046,10 @@ foreach i = 0-7 in { } let Predicates = [HasLS64] in { - let mayLoad = 1 in - def LD64B: LoadStore64B<0b101, "ld64b", (ins GPR64sp:$Rn), - (outs GPR64x8:$Rt)>; - let mayStore = 1 in - def ST64B: LoadStore64B<0b001, "st64b", (ins GPR64x8:$Rt, GPR64sp:$Rn), - (outs)>; + defm LD64B: LoadStore64B<0b101, "ld64b", 1, 0, (ins GPR64sp:$Rn), + (outs GPR64x8:$Rt)>; + defm ST64B: LoadStore64B<0b001, "st64b", 0, 1, (ins GPR64x8:$Rt, GPR64sp:$Rn), + (outs)>; def ST64BV: Store64BV<0b011, "st64bv">; def ST64BV0: Store64BV<0b010, "st64bv0">; @@ -11497,7 +12178,7 @@ defm UMIN : ComparisonOp<1, 1, "umin", umin>, Requires<[HasCSSC]>; def RPRFM: I<(outs), (ins rprfop:$Rt, GPR64:$Rm, GPR64sp:$Rn), "rprfm", "\t$Rt, $Rm, [$Rn]", "", []>, - Sched<[]> { + Sched<[WriteLD]> { bits<6> Rt; bits<5> Rn; bits<5> Rm; @@ -11921,13 +12602,6 @@ let Predicates = [HasF16F32MM] in let Uses = [FPMR, FPCR] in defm FMMLA : SIMDThreeSameVectorFP8MatrixMul<"fmmla", int_aarch64_neon_fmmla>; -//===----------------------------------------------------------------------===// -// Contention Management Hints (FEAT_CMH) -//===----------------------------------------------------------------------===// - -defm SHUH : SHUH<"shuh">; // Shared Update Hint instruction -def STCPH : STCPHInst<"stcph">; // Store Concurrent Priority Hint instruction - //===----------------------------------------------------------------------===// // Permission Overlays Extension 2 (FEAT_S1POE2) //===----------------------------------------------------------------------===// diff --git a/examples/llvm/Target/X86/X86.td b/examples/llvm/Target/X86/X86.td index c42d345..5786c65 100644 --- a/examples/llvm/Target/X86/X86.td +++ b/examples/llvm/Target/X86/X86.td @@ -44,8 +44,10 @@ def FeatureReserveEDI : SubtargetFeature<"reserve-edi", "ReservedRReg[X86::EDI]" def FeatureX87 : SubtargetFeature<"x87","HasX87", "true", "Enable X87 float instructions">; +// Does not have intrinsics or ABI effects. def FeatureNOPL : SubtargetFeature<"nopl", "HasNOPL", "true", - "Enable NOPL instruction (generally pentium pro+)">; + "Enable NOPL instruction (generally pentium pro+)", + [], InlineIgnore>; def FeatureCMOV : SubtargetFeature<"cmov","HasCMOV", "true", "Enable conditional move instructions">; @@ -104,7 +106,8 @@ def FeatureMMX : SubtargetFeature<"mmx","HasMMX", "true", // without disabling 64-bit mode. Nothing should imply this feature bit. It // is used to enforce that only 64-bit capable CPUs are used in 64-bit mode. def FeatureX86_64 : SubtargetFeature<"64bit", "HasX86_64", "true", - "Support 64-bit instructions">; + "Support 64-bit instructions", [], + InlineIgnore>; def FeatureCX16 : SubtargetFeature<"cx16", "HasCX16", "true", "64-bit with cmpxchg16b (this is true for most x86-64 chips, but not the first AMD chips)", [FeatureCX8]>; @@ -154,6 +157,9 @@ def FeatureVBMI : SubtargetFeature<"avx512vbmi", "HasVBMI", "true", def FeatureVBMI2 : SubtargetFeature<"avx512vbmi2", "HasVBMI2", "true", "Enable AVX-512 further Vector Byte Manipulation Instructions", [FeatureBWI]>; +def FeatureBMM : SubtargetFeature<"avx512bmm", "HasBMM", "true", + "Enable AVX512 Bit Matrix Multiply", + [FeatureBWI]>; def FeatureAVXIFMA : SubtargetFeature<"avxifma", "HasAVXIFMA", "true", "Enable AVX-IFMA", [FeatureAVX2]>; @@ -204,9 +210,10 @@ def FeatureFMA4 : SubtargetFeature<"fma4", "HasFMA4", "true", def FeatureXOP : SubtargetFeature<"xop", "HasXOP", "true", "Enable XOP instructions", [FeatureFMA4]>; -def FeatureSSEUnalignedMem : SubtargetFeature<"sse-unaligned-mem", - "HasSSEUnalignedMem", "true", - "Allow unaligned memory operands with SSE instructions (this may require setting a configuration bit in the processor)">; +def FeatureSSEUnalignedMem : SubtargetFeature< + "sse-unaligned-mem", "HasSSEUnalignedMem", "true", + "Allow unaligned memory operands with SSE instructions (this may require setting a configuration bit in the processor)", + [], InlineIgnore>; def FeatureAES : SubtargetFeature<"aes", "HasAES", "true", "Enable AES instructions", [FeatureSSE2]>; @@ -253,8 +260,10 @@ def FeaturePRFCHW : SubtargetFeature<"prfchw", "HasPRFCHW", "true", "Support PRFCHW instructions">; def FeatureRDSEED : SubtargetFeature<"rdseed", "HasRDSEED", "true", "Support RDSEED instruction">; +// Does not have intrinsics or ABI effects. def FeatureLAHFSAHF64 : SubtargetFeature<"sahf", "HasLAHFSAHF64", "true", - "Support LAHF and SAHF instructions in 64-bit mode">; + "Support LAHF and SAHF instructions in 64-bit mode", + [], InlineIgnore>; def FeatureMWAITX : SubtargetFeature<"mwaitx", "HasMWAITX", "true", "Enable MONITORX/MWAITX timer functionality">; def FeatureCLZERO : SubtargetFeature<"clzero", "HasCLZERO", "true", @@ -287,9 +296,6 @@ def FeatureAMXAVX512 : SubtargetFeature<"amx-avx512", "HasAMXAVX512", "true", "Support AMX-AVX512 instructions", [FeatureAMXTILE]>; -def FeatureAMXTF32 : SubtargetFeature<"amx-tf32", "HasAMXTF32", "true", - "Support AMX-TF32 instructions", - [FeatureAMXTILE]>; def FeatureCMPCCXADD : SubtargetFeature<"cmpccxadd", "HasCMPCCXADD", "true", "Support CMPCCXADD instructions">; def FeatureRAOINT : SubtargetFeature<"raoint", "HasRAOINT", "true", @@ -488,157 +494,232 @@ def FeatureHardenSlsIJmp //===----------------------------------------------------------------------===// def TuningPreferMovmskOverVTest : SubtargetFeature<"prefer-movmsk-over-vtest", "PreferMovmskOverVTest", "true", - "Prefer movmsk over vtest instruction">; + "Prefer movmsk over vtest instruction", + [], InlineIgnore>; def TuningSlowSHLD : SubtargetFeature<"slow-shld", "IsSHLDSlow", "true", - "SHLD instruction is slow">; + "SHLD instruction is slow", + [], InlineIgnore>; + +def TuningSlowPDEP : SubtargetFeature<"slow-pdep", "IsPDEPSlow", "true", + "PDEP instruction is slow", + [], InlineIgnore>; + +def TuningSlowPEXT : SubtargetFeature<"slow-pext", "IsPEXTSlow", "true", + "PEXT instruction is slow", + [], InlineIgnore>; def TuningSlowPMULLD : SubtargetFeature<"slow-pmulld", "IsPMULLDSlow", "true", - "PMULLD instruction is slow (compared to PMULLW/PMULHW and PMULUDQ)">; + "PMULLD instruction is slow (compared to PMULLW/PMULHW and PMULUDQ)", + [], InlineIgnore>; def TuningSlowPMULLQ : SubtargetFeature<"slow-pmullq", "IsPMULLQSlow", "true", - "PMULLQ instruction is slow">; + "PMULLQ instruction is slow", + [], InlineIgnore>; def TuningSlowPMADDWD : SubtargetFeature<"slow-pmaddwd", "IsPMADDWDSlow", "true", - "PMADDWD is slower than PMULLD">; + "PMADDWD is slower than PMULLD", + [], InlineIgnore>; + +def TuningSlowVecMaskStore + : SubtargetFeature<"slow-vec-mask-store", "IsVecMaskStoreSlow", "true", + "Vector mask store instruction is slow", + [], InlineIgnore>; // FIXME: This should not apply to CPUs that do not have SSE. def TuningSlowUAMem16 : SubtargetFeature<"slow-unaligned-mem-16", "IsUnalignedMem16Slow", "true", - "Slow unaligned 16-byte memory access">; + "Slow unaligned 16-byte memory access", + [], InlineIgnore>; def TuningSlowUAMem32 : SubtargetFeature<"slow-unaligned-mem-32", "IsUnalignedMem32Slow", "true", - "Slow unaligned 32-byte memory access">; + "Slow unaligned 32-byte memory access", + [], InlineIgnore>; def TuningLEAForSP : SubtargetFeature<"lea-sp", "UseLeaForSP", "true", - "Use LEA for adjusting the stack pointer (this is an optimization for Intel Atom processors)">; + "Use LEA for adjusting the stack pointer (this is an optimization for Intel Atom processors)", + [], InlineIgnore>; // True if 8-bit divisions are significantly faster than // 32-bit divisions and should be used when possible. def TuningSlowDivide32 : SubtargetFeature<"idivl-to-divb", "HasSlowDivide32", "true", - "Use 8-bit divide for positive values less than 256">; + "Use 8-bit divide for positive values less than 256", + [], InlineIgnore>; // True if 32-bit divides are significantly faster than // 64-bit divisions and should be used when possible. def TuningSlowDivide64 : SubtargetFeature<"idivq-to-divl", "HasSlowDivide64", "true", - "Use 32-bit divide for positive values less than 2^32">; + "Use 32-bit divide for positive values less than 2^32", + [], InlineIgnore>; def TuningPadShortFunctions : SubtargetFeature<"pad-short-functions", "PadShortFunctions", "true", - "Pad short functions (to prevent a stall when returning too early)">; + "Pad short functions (to prevent a stall when returning too early)", + [], InlineIgnore>; // On some processors, instructions that implicitly take two memory operands are // slow. In practice, this means that CALL, PUSH, and POP with memory operands // should be avoided in favor of a MOV + register CALL/PUSH/POP. def TuningSlowTwoMemOps : SubtargetFeature<"slow-two-mem-ops", "SlowTwoMemOps", "true", - "Two memory operand instructions are slow">; + "Two memory operand instructions are slow", + [], InlineIgnore>; + +// On some processors, indirect calls from memory (CALL [mem]) are slow +// compared to loading the address first and using a register indirect call. +// This avoids folding loads into indirect call instructions, but does not +// affect PUSH from memory folding. +def TuningSlowIndirectCall : SubtargetFeature<"slow-indirect-call", + "SlowIndirectCall", "true", + "Indirect calls from memory are slow", + [], InlineIgnore>; // True if the LEA instruction inputs have to be ready at address generation // (AG) time. def TuningLEAUsesAG : SubtargetFeature<"lea-uses-ag", "LeaUsesAG", "true", - "LEA instruction needs inputs at AG stage">; + "LEA instruction needs inputs at AG stage", + [], InlineIgnore>; def TuningSlowLEA : SubtargetFeature<"slow-lea", "SlowLEA", "true", - "LEA instruction with certain arguments is slow">; + "LEA instruction with certain arguments is slow", + [], InlineIgnore>; // True if the LEA instruction has all three source operands: base, index, // and offset or if the LEA instruction uses base and index registers where // the base is EBP, RBP,or R13 def TuningSlow3OpsLEA : SubtargetFeature<"slow-3ops-lea", "Slow3OpsLEA", "true", - "LEA instruction with 3 ops or certain registers is slow">; + "LEA instruction with 3 ops or certain registers is slow", + [], InlineIgnore>; // True if INC and DEC instructions are slow when writing to flags def TuningSlowIncDec : SubtargetFeature<"slow-incdec", "SlowIncDec", "true", - "INC and DEC instructions are slower than ADD and SUB">; + "INC and DEC instructions are slower than ADD and SUB", + [], InlineIgnore>; def TuningPOPCNTFalseDeps : SubtargetFeature<"false-deps-popcnt", "HasPOPCNTFalseDeps", "true", - "POPCNT has a false dependency on dest register">; + "POPCNT has a false dependency on dest register", + [], InlineIgnore>; -def TuningLZCNTFalseDeps : SubtargetFeature<"false-deps-lzcnt-tzcnt", +def TuningLZCNTFalseDeps : SubtargetFeature<"false-deps-lzcnt", "HasLZCNTFalseDeps", "true", - "LZCNT/TZCNT have a false dependency on dest register">; + "LZCNT has a false dependency on dest register", + [], InlineIgnore>; + +def TuningTZCNTFalseDeps : SubtargetFeature<"false-deps-tzcnt", + "HasTZCNTFalseDeps", "true", + "TZCNT has a false dependency on dest register", + [], InlineIgnore>; + +def TuningBLSFalseDeps : SubtargetFeature<"false-deps-bls", + "HasBLSFalseDeps", "true", + "BLSR/BLSI/BLSMSK have a false dependency on dest register", + [], InlineIgnore>; def TuningMULCFalseDeps : SubtargetFeature<"false-deps-mulc", "HasMULCFalseDeps", "true", - "VF[C]MULCPH/SH has a false dependency on dest register">; + "VF[C]MULCPH/SH has a false dependency on dest register", + [], InlineIgnore>; def TuningPERMFalseDeps : SubtargetFeature<"false-deps-perm", "HasPERMFalseDeps", "true", - "VPERMD/Q/PS/PD has a false dependency on dest register">; + "VPERMD/Q/PS/PD has a false dependency on dest register", + [], InlineIgnore>; + +def TuningCOMPRESSFalseDeps : SubtargetFeature<"false-deps-compress", + "HasCOMPRESSFalseDeps", "true", + "V[P]COMPRESSPD/PS/Q/D/W/B with zero-masking has a false dependency on dest register", + [], InlineIgnore>; + +def TuningEXPANDFalseDeps : SubtargetFeature<"false-deps-expand", + "HasEXPANDFalseDeps", "true", + "V[P]EXPANDPD/PS/Q/D/W/B with zero-masking has a false dependency on dest register", + [], InlineIgnore>; def TuningRANGEFalseDeps : SubtargetFeature<"false-deps-range", "HasRANGEFalseDeps", "true", - "VRANGEPD/PS/SD/SS has a false dependency on dest register">; + "VRANGEPD/PS/SD/SS has a false dependency on dest register", + [], InlineIgnore>; def TuningGETMANTFalseDeps : SubtargetFeature<"false-deps-getmant", "HasGETMANTFalseDeps", "true", "VGETMANTSS/SD/SH and VGETMANDPS/PD(memory version) has a" - " false dependency on dest register">; + " false dependency on dest register", + [], InlineIgnore>; def TuningMULLQFalseDeps : SubtargetFeature<"false-deps-mullq", "HasMULLQFalseDeps", "true", - "VPMULLQ has a false dependency on dest register">; + "VPMULLQ has a false dependency on dest register", + [], InlineIgnore>; def TuningSBBDepBreaking : SubtargetFeature<"sbb-dep-breaking", "HasSBBDepBreaking", "true", - "SBB with same register has no source dependency">; + "SBB with same register has no source dependency", + [], InlineIgnore>; // On recent X86 (port bound) processors, its preferable to combine to a single shuffle // using a variable mask over multiple fixed shuffles. def TuningFastVariableCrossLaneShuffle : SubtargetFeature<"fast-variable-crosslane-shuffle", "HasFastVariableCrossLaneShuffle", - "true", "Cross-lane shuffles with variable masks are fast">; + "true", "Cross-lane shuffles with variable masks are fast", + [], InlineIgnore>; def TuningFastVariablePerLaneShuffle : SubtargetFeature<"fast-variable-perlane-shuffle", "HasFastVariablePerLaneShuffle", - "true", "Per-lane shuffles with variable masks are fast">; + "true", "Per-lane shuffles with variable masks are fast", + [], InlineIgnore>; // Goldmont / Tremont (atom in general) has no bypass delay def TuningNoDomainDelay : SubtargetFeature<"no-bypass-delay", "NoDomainDelay","true", - "Has no bypass delay when using the 'wrong' domain">; + "Has no bypass delay when using the 'wrong' domain", + [], InlineIgnore>; // Many processors (Nehalem+ on Intel) have no bypass delay when // using the wrong mov type. def TuningNoDomainDelayMov : SubtargetFeature<"no-bypass-delay-mov", "NoDomainDelayMov","true", - "Has no bypass delay when using the 'wrong' mov type">; + "Has no bypass delay when using the 'wrong' mov type", + [], InlineIgnore>; // Newer processors (Skylake+ on Intel) have no bypass delay when // using the wrong blend type. def TuningNoDomainDelayBlend : SubtargetFeature<"no-bypass-delay-blend", "NoDomainDelayBlend","true", - "Has no bypass delay when using the 'wrong' blend type">; + "Has no bypass delay when using the 'wrong' blend type", + [], InlineIgnore>; // Newer processors (Haswell+ on Intel) have no bypass delay when // using the wrong shuffle type. def TuningNoDomainDelayShuffle : SubtargetFeature<"no-bypass-delay-shuffle", "NoDomainDelayShuffle","true", - "Has no bypass delay when using the 'wrong' shuffle type">; + "Has no bypass delay when using the 'wrong' shuffle type", + [], InlineIgnore>; // Prefer lowering shuffles on AVX512 targets (e.g. Skylake Server) to // imm shifts/rotate if they can use more ports than regular shuffles. def TuningPreferShiftShuffle : SubtargetFeature<"faster-shift-than-shuffle", "PreferLowerShuffleAsShift", "true", - "Shifts are faster (or as fast) as shuffle">; + "Shifts are faster (or as fast) as shuffle", + [], InlineIgnore>; def TuningFastImmVectorShift : SubtargetFeature<"tuning-fast-imm-vector-shift", "FastImmVectorShift", "true", - "Vector shifts are fast (2/cycle) as opposed to slow (1/cycle)">; + "Vector shifts are fast (2/cycle) as opposed to slow (1/cycle)", + [], InlineIgnore>; // On some X86 processors, a vzeroupper instruction should be inserted after // using ymm/zmm registers before executing code that may use SSE instructions. def TuningInsertVZEROUPPER : SubtargetFeature<"vzeroupper", "InsertVZEROUPPER", - "true", "Should insert vzeroupper instructions">; + "true", "Should insert vzeroupper instructions", + [], InlineIgnore>; // TuningFastScalarFSQRT should be enabled if scalar FSQRT has shorter latency // than the corresponding NR code. TuningFastVectorFSQRT should be enabled if @@ -652,37 +733,43 @@ def TuningInsertVZEROUPPER // RSQRTSS followed by a Newton-Raphson iteration. def TuningFastScalarFSQRT : SubtargetFeature<"fast-scalar-fsqrt", "HasFastScalarFSQRT", - "true", "Scalar SQRT is fast (disable Newton-Raphson)">; + "true", "Scalar SQRT is fast (disable Newton-Raphson)", + [], InlineIgnore>; // True if hardware SQRTPS/VSQRTPS instructions are at least as fast // (throughput) as RSQRTPS/VRSQRTPS followed by a Newton-Raphson iteration. def TuningFastVectorFSQRT : SubtargetFeature<"fast-vector-fsqrt", "HasFastVectorFSQRT", - "true", "Vector SQRT is fast (disable Newton-Raphson)">; + "true", "Vector SQRT is fast (disable Newton-Raphson)", + [], InlineIgnore>; // If lzcnt has equivalent latency/throughput to most simple integer ops, it can // be used to replace test/set sequences. def TuningFastLZCNT : SubtargetFeature< "fast-lzcnt", "HasFastLZCNT", "true", - "LZCNT instructions are as fast as most simple integer ops">; + "LZCNT instructions are as fast as most simple integer ops", + [], InlineIgnore>; // If the target can efficiently decode NOPs upto 7-bytes in length. def TuningFast7ByteNOP : SubtargetFeature< "fast-7bytenop", "HasFast7ByteNOP", "true", - "Target can quickly decode up to 7 byte NOPs">; + "Target can quickly decode up to 7 byte NOPs", + [], InlineIgnore>; // If the target can efficiently decode NOPs upto 11-bytes in length. def TuningFast11ByteNOP : SubtargetFeature< "fast-11bytenop", "HasFast11ByteNOP", "true", - "Target can quickly decode up to 11 byte NOPs">; + "Target can quickly decode up to 11 byte NOPs", + [], InlineIgnore>; // If the target can efficiently decode NOPs upto 15-bytes in length. def TuningFast15ByteNOP : SubtargetFeature< "fast-15bytenop", "HasFast15ByteNOP", "true", - "Target can quickly decode up to 15 byte NOPs">; + "Target can quickly decode up to 15 byte NOPs", + [], InlineIgnore>; // Sandy Bridge and newer processors can use SHLD with the same source on both // inputs to implement rotate to avoid the partial flag update of the normal @@ -690,20 +777,23 @@ def TuningFast15ByteNOP def TuningFastSHLDRotate : SubtargetFeature< "fast-shld-rotate", "HasFastSHLDRotate", "true", - "SHLD can be used as a faster rotate">; + "SHLD can be used as a faster rotate", + [], InlineIgnore>; // Bulldozer and newer processors can merge CMP/TEST (but not other // instructions) with conditional branches. def TuningBranchFusion : SubtargetFeature<"branchfusion", "HasBranchFusion", "true", - "CMP/TEST can be fused with conditional branches">; + "CMP/TEST can be fused with conditional branches", + [], InlineIgnore>; // Sandy Bridge and newer processors have many instructions that can be // fused with conditional branches and pass through the CPU as a single // operation. def TuningMacroFusion : SubtargetFeature<"macrofusion", "HasMacroFusion", "true", - "Various instructions can be fused with conditional branches">; + "Various instructions can be fused with conditional branches", + [], InlineIgnore>; // Gather is available since Haswell (AVX2 set). So technically, we can // generate Gathers on all AVX2 processors. But the overhead on HSW is high. @@ -711,40 +801,49 @@ def TuningMacroFusion // similar to Skylake Server (AVX-512). def TuningFastGather : SubtargetFeature<"fast-gather", "HasFastGather", "true", - "Indicates if gather is reasonably fast (this is true for Skylake client and all AVX-512 CPUs)">; + "Indicates if gather is reasonably fast (this is true for Skylake client and all AVX-512 CPUs)", + [], InlineIgnore>; // Generate vpdpwssd instead of vpmaddwd+vpaddd sequence. def TuningFastDPWSSD : SubtargetFeature< "fast-dpwssd", "HasFastDPWSSD", "true", - "Prefer vpdpwssd instruction over vpmaddwd+vpaddd instruction sequence">; + "Prefer vpdpwssd instruction over vpmaddwd+vpaddd instruction sequence", + [], InlineIgnore>; def TuningPreferNoGather : SubtargetFeature<"prefer-no-gather", "PreferGather", "false", - "Prefer no gather instructions">; + "Prefer no gather instructions", + [], InlineIgnore>; def TuningPreferNoScatter : SubtargetFeature<"prefer-no-scatter", "PreferScatter", "false", - "Prefer no scatter instructions">; + "Prefer no scatter instructions", + [], InlineIgnore>; def TuningPrefer128Bit : SubtargetFeature<"prefer-128-bit", "Prefer128Bit", "true", - "Prefer 128-bit AVX instructions">; + "Prefer 128-bit AVX instructions", + [], InlineIgnore>; def TuningPrefer256Bit : SubtargetFeature<"prefer-256-bit", "Prefer256Bit", "true", - "Prefer 256-bit AVX instructions">; + "Prefer 256-bit AVX instructions", + [], InlineIgnore>; def TuningAllowLight256Bit : SubtargetFeature<"allow-light-256-bit", "AllowLight256Bit", "true", - "Enable generation of 256-bit load/stores even if we prefer 128-bit">; + "Enable generation of 256-bit load/stores even if we prefer 128-bit", + [], InlineIgnore>; def TuningPreferMaskRegisters : SubtargetFeature<"prefer-mask-registers", "PreferMaskRegisters", "true", - "Prefer AVX512 mask registers over PTEST/MOVMSK">; + "Prefer AVX512 mask registers over PTEST/MOVMSK", + [], InlineIgnore>; def TuningFastBEXTR : SubtargetFeature<"fast-bextr", "HasFastBEXTR", "true", "Indicates that the BEXTR instruction is implemented as a single uop " - "with good throughput">; + "with good throughput", + [], InlineIgnore>; // Combine vector math operations with shuffles into horizontal math // instructions if a CPU implements horizontal operations (introduced with @@ -753,50 +852,53 @@ def TuningFastHorizontalOps : SubtargetFeature< "fast-hops", "HasFastHorizontalOps", "true", "Prefer horizontal vector math instructions (haddp, phsub, etc.) over " - "normal vector instructions with shuffles">; + "normal vector instructions with shuffles", + [], InlineIgnore>; def TuningFastScalarShiftMasks : SubtargetFeature< "fast-scalar-shift-masks", "HasFastScalarShiftMasks", "true", - "Prefer a left/right scalar logical shift pair over a shift+and pair">; + "Prefer a left/right scalar logical shift pair over a shift+and pair", + [], InlineIgnore>; def TuningFastVectorShiftMasks : SubtargetFeature< "fast-vector-shift-masks", "HasFastVectorShiftMasks", "true", - "Prefer a left/right vector logical shift pair over a shift+and pair">; + "Prefer a left/right vector logical shift pair over a shift+and pair", + [], InlineIgnore>; def TuningFastMOVBE : SubtargetFeature<"fast-movbe", "HasFastMOVBE", "true", - "Prefer a movbe over a single-use load + bswap / single-use bswap + store">; + "Prefer a movbe over a single-use load + bswap / single-use bswap + store", + [], InlineIgnore>; def TuningFastImm16 : SubtargetFeature<"fast-imm16", "HasFastImm16", "true", - "Prefer a i16 instruction with i16 immediate over extension to i32">; + "Prefer a i16 instruction with i16 immediate over extension to i32", + [], InlineIgnore>; def TuningUseSLMArithCosts : SubtargetFeature<"use-slm-arith-costs", "UseSLMArithCosts", "true", - "Use Silvermont specific arithmetic costs">; + "Use Silvermont specific arithmetic costs", [], InlineIgnore>; def TuningUseGLMDivSqrtCosts : SubtargetFeature<"use-glm-div-sqrt-costs", "UseGLMDivSqrtCosts", "true", - "Use Goldmont specific floating point div/sqrt costs">; - -def TuningNDDImm - : SubtargetFeature<"prefer-ndd-imm", "HasNDDI", - "true", "Prefer NDD immediate variant">; + "Use Goldmont specific floating point div/sqrt costs", + [], InlineIgnore>; def TuningNDDMem : SubtargetFeature<"prefer-ndd-mem", "HasNDDM", - "true", "Prefer NDD memory addressing">; + "true", "Prefer NDD memory addressing", [], InlineIgnore>; // Starting with Redwood Cove architecture, the branch has branch taken hint // (i.e., instruction prefix 3EH). def TuningBranchHint: SubtargetFeature<"branch-hint", "HasBranchHint", "true", - "Target has branch hint feature">; + "Target has branch hint feature", + [], InlineIgnore>; def TuningPreferLegacySetCC : SubtargetFeature<"prefer-legacy-setcc", "PreferLegacySetCC", "true", - "Prefer to emit legacy SetCC.">; + "Prefer to emit legacy SetCC.", [], InlineIgnore>; //===----------------------------------------------------------------------===// // X86 CPU Families @@ -804,7 +906,7 @@ def TuningPreferLegacySetCC //===----------------------------------------------------------------------===// // Bonnell -def ProcIntelAtom : SubtargetFeature<"", "IsAtom", "true", "Is Intel Atom processor">; +def ProcIntelAtom : SubtargetFeature<"", "IsAtom", "true", "Is Intel Atom processor", [], InlineIgnore>; //===----------------------------------------------------------------------===// // Register File Description @@ -846,6 +948,10 @@ include "X86SchedIceLake.td" include "X86SchedAlderlakeP.td" include "X86SchedLunarlakeP.td" include "X86SchedSapphireRapids.td" +include "X86ScheduleC864GM4.td" +include "X86ScheduleC864GM7.td" +include "X86ScheduleC864GM8.td" + //===----------------------------------------------------------------------===// // X86 Processor Feature Lists @@ -890,14 +996,15 @@ def ProcessorFeatures { TuningSlow3OpsLEA, TuningSlowDivide64, TuningFastScalarFSQRT, - TuningFastSHLDRotate, TuningFast15ByteNOP, TuningFastVariableCrossLaneShuffle, TuningFastVariablePerLaneShuffle, TuningPOPCNTFalseDeps, TuningLZCNTFalseDeps, + TuningTZCNTFalseDeps, TuningInsertVZEROUPPER, - TuningAllowLight256Bit + TuningAllowLight256Bit, + TuningSlowVecMaskStore ]; list X86_64V4Features = !listconcat(X86_64V3Features, [ @@ -912,9 +1019,7 @@ def ProcessorFeatures { TuningSlowDivide64, TuningFastScalarFSQRT, TuningFastVectorFSQRT, - TuningFastSHLDRotate, TuningFast15ByteNOP, - TuningFastVariableCrossLaneShuffle, TuningFastVariablePerLaneShuffle, TuningPrefer256Bit, TuningFastGather, @@ -974,12 +1079,12 @@ def ProcessorFeatures { TuningSlow3OpsLEA, TuningSlowDivide64, TuningFastScalarFSQRT, - TuningFastSHLDRotate, TuningFast15ByteNOP, TuningFastVariableCrossLaneShuffle, TuningFastVariablePerLaneShuffle, TuningPOPCNTFalseDeps, TuningLZCNTFalseDeps, + TuningTZCNTFalseDeps, TuningInsertVZEROUPPER, TuningAllowLight256Bit, TuningNoDomainDelayMov, @@ -1006,7 +1111,6 @@ def ProcessorFeatures { TuningSlowDivide64, TuningFastScalarFSQRT, TuningFastVectorFSQRT, - TuningFastSHLDRotate, TuningFast15ByteNOP, TuningFastVariableCrossLaneShuffle, TuningFastVariablePerLaneShuffle, @@ -1037,7 +1141,6 @@ def ProcessorFeatures { TuningSlowDivide64, TuningFastScalarFSQRT, TuningFastVectorFSQRT, - TuningFastSHLDRotate, TuningFast15ByteNOP, TuningFastVariableCrossLaneShuffle, TuningFastVariablePerLaneShuffle, @@ -1081,7 +1184,6 @@ def ProcessorFeatures { TuningSlowDivide64, TuningFastScalarFSQRT, TuningFastVectorFSQRT, - TuningFastSHLDRotate, TuningFast15ByteNOP, TuningFastVariableCrossLaneShuffle, TuningFastVariablePerLaneShuffle, @@ -1111,7 +1213,6 @@ def ProcessorFeatures { TuningSlowDivide64, TuningFastScalarFSQRT, TuningFastVectorFSQRT, - TuningFastSHLDRotate, TuningFast15ByteNOP, TuningFastVariableCrossLaneShuffle, TuningFastVariablePerLaneShuffle, @@ -1204,10 +1305,11 @@ def ProcessorFeatures { FeatureMOVRS, FeatureAMXMOVRS, FeatureAMXAVX512, - FeatureAMXFP8, - FeatureAMXTF32]; + FeatureAMXFP8]; list DMRFeatures = !listconcat(GNRDFeatures, DMRAdditionalFeatures); + list DMRAdditionalTuning = [TuningPreferLegacySetCC]; + list DMRTuning = !listconcat(GNRTuning, DMRAdditionalTuning); // Atom list AtomFeatures = [FeatureX87, @@ -1379,6 +1481,8 @@ def ProcessorFeatures { FeaturePREFETCHI]; list NVLFeatures = !listconcat(PTLFeatures, NVLAdditionalFeatures); + list NVLAdditionalTuning = [TuningPreferLegacySetCC]; + list NVLTuning = !listconcat(ADLTuning, NVLAdditionalTuning); // Clearwaterforest list CWFAdditionalFeatures = [FeaturePREFETCHI, @@ -1428,7 +1532,6 @@ def ProcessorFeatures { TuningFastMOVBE, TuningFastImm16, TuningSlowPMADDWD]; - // TODO Add AVX5124FMAPS/AVX5124VNNIW features list KNMFeatures = !listconcat(KNLFeatures, [FeatureVPOPCNTDQ]); @@ -1551,7 +1654,10 @@ def ProcessorFeatures { FeatureMOVBE, FeatureRDRAND, FeatureMWAITX]; - list BdVer4Tuning = BdVer3Tuning; + list BdVer4AdditionalTuning = [TuningSlowPDEP, + TuningSlowPEXT]; + list BdVer4Tuning = + !listconcat(BdVer3Tuning, BdVer4AdditionalTuning); list BdVer4Features = !listconcat(BdVer3Features, BdVer4AdditionalFeatures); @@ -1601,10 +1707,13 @@ def ProcessorFeatures { TuningFastMOVBE, TuningFastImm16, TuningSlowDivide64, + TuningSlowPDEP, + TuningSlowPEXT, TuningSlowSHLD, TuningSBBDepBreaking, TuningInsertVZEROUPPER, - TuningAllowLight256Bit]; + TuningAllowLight256Bit, + TuningSlowVecMaskStore]; list ZN2AdditionalFeatures = [FeatureCLWB, FeatureRDPID, FeatureRDPRU, @@ -1618,14 +1727,20 @@ def ProcessorFeatures { FeatureVAES, FeatureVPCLMULQDQ]; list ZN3AdditionalTuning = [TuningMacroFusion]; + list ZN3RemoveTuning = [TuningSlowPDEP, + TuningSlowPEXT, + TuningSlowSHLD]; list ZN3Tuning = - !listremove(!listconcat(ZN2Tuning, ZN3AdditionalTuning), [TuningSlowSHLD]); + !listremove(!listconcat(ZN2Tuning, ZN3AdditionalTuning), ZN3RemoveTuning); list ZN3Features = !listconcat(ZN2Features, ZN3AdditionalFeatures); - list ZN4AdditionalTuning = [TuningFastDPWSSD]; + list ZN4AdditionalTuning = [TuningFastDPWSSD, + TuningCOMPRESSFalseDeps, + TuningEXPANDFalseDeps]; list ZN4Tuning = - !listconcat(ZN3Tuning, ZN4AdditionalTuning); + !listremove(!listconcat(ZN3Tuning, ZN4AdditionalTuning), + [TuningSlowVecMaskStore]); list ZN4AdditionalFeatures = [FeatureAVX512, FeatureCDI, FeatureDQI, @@ -1643,7 +1758,12 @@ def ProcessorFeatures { list ZN4Features = !listconcat(ZN3Features, ZN4AdditionalFeatures); - list ZN5Tuning = ZN4Tuning; + list ZN5AdditionalTuning = [TuningSlowIndirectCall, + TuningTZCNTFalseDeps, + TuningBLSFalseDeps + ]; + list ZN5Tuning = + !listconcat(ZN4Tuning, ZN5AdditionalTuning); list ZN5AdditionalFeatures = [FeatureMOVDIRI, FeatureMOVDIR64B, FeatureVP2INTERSECT, @@ -1653,14 +1773,107 @@ def ProcessorFeatures { list ZN5Features = !listconcat(ZN4Features, ZN5AdditionalFeatures); - list ZN6Tuning = ZN5Tuning; + list ZN6RemoveTuning = [TuningSlowIndirectCall, + TuningTZCNTFalseDeps, + TuningBLSFalseDeps + ]; + list ZN6Tuning = + !listremove(ZN5Tuning, ZN6RemoveTuning); list ZN6AdditionalFeatures = [FeatureFP16, FeatureAVXVNNIINT8, FeatureAVXNECONVERT, - FeatureAVXIFMA + FeatureAVXIFMA, + FeatureBMM ]; list ZN6Features = !listconcat(ZN5Features, ZN6AdditionalFeatures); + + // Hygon Processors common ISAs + list C864GM4Features = [FeatureADX, + FeatureAES, + FeatureAVX2, + FeatureBMI, + FeatureBMI2, + FeatureCLFLUSHOPT, + FeatureCLZERO, + FeatureCMOV, + FeatureCRC32, + FeatureCX16, + FeatureF16C, + FeatureFMA, + FeatureFSGSBase, + FeatureFXSR, + FeatureLAHFSAHF64, + FeatureLZCNT, + FeatureMMX, + FeatureMOVBE, + FeatureMWAITX, + FeatureNOPL, + FeaturePCLMUL, + FeaturePOPCNT, + FeaturePRFCHW, + FeatureRDRAND, + FeatureRDSEED, + FeatureSHA, + FeatureSSE4A, + FeatureX86_64, + FeatureX87, + FeatureXSAVE, + FeatureXSAVEC, + FeatureXSAVEOPT, + FeatureXSAVES]; + list C864GM4Tuning = [TuningFastLZCNT, + TuningFastBEXTR, + TuningFast15ByteNOP, + TuningFastScalarFSQRT, + TuningFastVectorFSQRT, + TuningFastScalarShiftMasks, + TuningFastVariablePerLaneShuffle, + TuningFastMOVBE, + TuningFastImm16, + TuningSlowDivide64, + TuningSlowSHLD, + TuningSBBDepBreaking, + TuningInsertVZEROUPPER, + TuningAllowLight256Bit]; + + // C86-4G-M6: same features as M4; kept separate for future differentiation + list C864GM6Features = C864GM4Features; + list C864GM6Tuning = C864GM4Tuning; + + // C86-4G-M7 + list C864GM7AdditionalFeatures = [FeatureAVX512, + FeatureBF16, + FeatureBITALG, + FeatureBWI, + FeatureCDI, + FeatureCLWB, + FeatureDQI, + FeatureGFNI, + FeatureIFMA, + FeatureVAES, + FeatureVBMI, + FeatureVBMI2, + FeatureVLX, + FeatureVNNI, + FeatureVPCLMULQDQ, + FeatureVPOPCNTDQ, + FeatureWBNOINVD]; + list C864GM7Features = + !listconcat(C864GM4Features, C864GM7AdditionalFeatures); + + list C864GM7AdditionalTuning = [TuningBranchFusion, + TuningPreferNoGather, + TuningPreferNoScatter, + TuningPrefer256Bit]; + list C864GM7Tuning = + !listconcat(C864GM4Tuning, C864GM7AdditionalTuning); + + // C86-4G-M8 + list C864GM8AdditionalFeatures = [FeatureSHSTK]; + list C864GM8Features = + !listconcat(C864GM7Features, C864GM8AdditionalFeatures); + list C864GM8Tuning = C864GM7Tuning; } //===----------------------------------------------------------------------===// @@ -1699,28 +1912,26 @@ def : Proc<"i486", [FeatureX87], [TuningSlowUAMem16, TuningInsertVZEROUPPER]>; def : Proc<"i586", [FeatureX87, FeatureCX8], [TuningSlowUAMem16, TuningInsertVZEROUPPER]>; -def : Proc<"pentium", [FeatureX87, FeatureCX8], - [TuningSlowUAMem16, TuningInsertVZEROUPPER]>; -foreach P = ["pentium-mmx", "pentium_mmx"] in { - def : Proc; -} +def : ProcessorAlias<"pentium", "i586">; + +def : Proc<"pentium-mmx", [FeatureX87, FeatureCX8, FeatureMMX], + [TuningSlowUAMem16, TuningInsertVZEROUPPER]>; +def : ProcessorAlias<"pentium_mmx", "pentium-mmx">; def : Proc<"i686", [FeatureX87, FeatureCX8, FeatureCMOV], [TuningSlowUAMem16, TuningInsertVZEROUPPER]>; -foreach P = ["pentiumpro", "pentium_pro"] in { - def : Proc; -} -foreach P = ["pentium2", "pentium_ii"] in { - def : Proc; -} -foreach P = ["pentium3", "pentium3m", "pentium_iii_no_xmm_regs", "pentium_iii"] in { - def : Proc; -} +def : Proc<"pentiumpro", [FeatureX87, FeatureCX8, FeatureCMOV, FeatureNOPL], + [TuningSlowUAMem16, TuningInsertVZEROUPPER]>; +def : ProcessorAlias<"pentium_pro", "pentiumpro">; +def : Proc<"pentium2", [FeatureX87, FeatureCX8, FeatureMMX, FeatureCMOV, + FeatureFXSR, FeatureNOPL], + [TuningSlowUAMem16, TuningInsertVZEROUPPER]>; +def : ProcessorAlias<"pentium_ii", "pentium2">; +def : Proc<"pentium3", [FeatureX87, FeatureCX8, FeatureMMX, + FeatureSSE1, FeatureFXSR, FeatureNOPL, FeatureCMOV], + [TuningSlowUAMem16, TuningInsertVZEROUPPER]>; +def : ProcessorAlias<"pentium3m", "pentium3">; +def : ProcessorAlias<"pentium_iii_no_xmm_regs", "pentium3">; +def : ProcessorAlias<"pentium_iii", "pentium3">; // Enable the PostRAScheduler for SSE2 and SSE3 class cpus. // The intent is to enable it for pentium4 which is the current default @@ -1732,19 +1943,18 @@ foreach P = ["pentium3", "pentium3m", "pentium_iii_no_xmm_regs", "pentium_iii"] // measure to avoid performance surprises, in case clang's default cpu // changes slightly. -foreach P = ["pentium_m", "pentium-m"] in { -def : ProcModel; -} +def : ProcessorAlias<"pentium-m", "pentium_m">; -foreach P = ["pentium4", "pentium4m", "pentium_4"] in { - def : ProcModel; -} +def : ProcModel<"pentium4", GenericPostRAModel, + [FeatureX87, FeatureCX8, FeatureMMX, FeatureSSE2, + FeatureFXSR, FeatureNOPL, FeatureCMOV], + [TuningSlowUAMem16, TuningInsertVZEROUPPER]>; +def : ProcessorAlias<"pentium4m", "pentium4">; +def : ProcessorAlias<"pentium_4", "pentium4">; // Intel Quark. def : Proc<"lakemont", [FeatureCX8], @@ -1757,12 +1967,11 @@ def : ProcModel<"yonah", SandyBridgeModel, [TuningSlowUAMem16, TuningInsertVZEROUPPER]>; // NetBurst. -foreach P = ["prescott", "pentium_4_sse3"] in { - def : ProcModel; -} +def : ProcModel<"prescott", GenericPostRAModel, + [FeatureX87, FeatureCX8, FeatureMMX, FeatureSSE3, + FeatureFXSR, FeatureNOPL, FeatureCMOV], + [TuningSlowUAMem16, TuningInsertVZEROUPPER]>; +def : ProcessorAlias<"pentium_4_sse3", "prescott">; def : ProcModel<"nocona", GenericPostRAModel, [ FeatureX87, FeatureCX8, @@ -1780,8 +1989,7 @@ def : ProcModel<"nocona", GenericPostRAModel, [ ]>; // Intel Core 2 Solo/Duo. -foreach P = ["core2", "core_2_duo_ssse3"] in { -def : ProcModel; -} -foreach P = ["penryn", "core_2_duo_sse4_1"] in { -def : ProcModel; +def : ProcModel<"penryn", SandyBridgeModel, [ FeatureX87, FeatureCX8, FeatureCMOV, @@ -1817,77 +2024,74 @@ def : ProcModel; -} +def : ProcessorAlias<"core_2_duo_sse4_1", "penryn">; // Atom CPUs. -foreach P = ["bonnell", "atom"] in { - def : ProcModel; -} +def : ProcModel<"bonnell", AtomModel, ProcessorFeatures.AtomFeatures, + ProcessorFeatures.AtomTuning>; +def : ProcessorAlias<"atom", "bonnell">; -foreach P = ["silvermont", "slm", "atom_sse4_2"] in { - def : ProcModel; -} +def : ProcModel<"silvermont", SLMModel, ProcessorFeatures.SLMFeatures, + ProcessorFeatures.SLMTuning>; +def : ProcessorAlias<"slm", "silvermont">; +def : ProcessorAlias<"atom_sse4_2", "silvermont">; def : ProcModel<"atom_sse4_2_movbe", SLMModel, ProcessorFeatures.GLMFeatures, ProcessorFeatures.SLMTuning>; def : ProcModel<"goldmont", SLMModel, ProcessorFeatures.GLMFeatures, ProcessorFeatures.GLMTuning>; -foreach P = ["goldmont_plus", "goldmont-plus"] in { - def : ProcModel; -} +def : ProcModel<"goldmont_plus", SLMModel, ProcessorFeatures.GLPFeatures, + ProcessorFeatures.GLPTuning>; +def : ProcessorAlias<"goldmont-plus", "goldmont_plus">; def : ProcModel<"tremont", SLMModel, ProcessorFeatures.TRMFeatures, ProcessorFeatures.TRMTuning>; // "Arrandale" along with corei3 and corei5 -foreach P = ["nehalem", "corei7", "core_i7_sse4_2"] in { - def : ProcModel; -} +def : ProcModel<"nehalem", SandyBridgeModel, ProcessorFeatures.NHMFeatures, + ProcessorFeatures.NHMTuning>; +def : ProcessorAlias<"corei7", "nehalem">; +def : ProcessorAlias<"core_i7_sse4_2", "nehalem">; // Westmere is the corei3/i5/i7 path from nehalem to sandybridge -foreach P = ["westmere", "core_aes_pclmulqdq"] in { - def : ProcModel; -} - -foreach P = ["sandybridge", "corei7-avx", "core_2nd_gen_avx"] in { - def : ProcModel; -} - -foreach P = ["ivybridge", "core-avx-i", "core_3rd_gen_avx"] in { - def : ProcModel; -} - -foreach P = ["haswell", "core-avx2", "core_4th_gen_avx", "core_4th_gen_avx_tsx"] in { - def : ProcModel; -} - -foreach P = ["broadwell", "core_5th_gen_avx", "core_5th_gen_avx_tsx"] in { - def : ProcModel; -} +def : ProcModel<"westmere", SandyBridgeModel, ProcessorFeatures.WSMFeatures, + ProcessorFeatures.WSMTuning>; +def : ProcessorAlias<"core_aes_pclmulqdq", "westmere">; + +def : ProcModel<"sandybridge", SandyBridgeModel, ProcessorFeatures.SNBFeatures, + ProcessorFeatures.SNBTuning>; +def : ProcessorAlias<"corei7-avx", "sandybridge">; +def : ProcessorAlias<"core_2nd_gen_avx", "sandybridge">; + +def : ProcModel<"ivybridge", SandyBridgeModel, ProcessorFeatures.IVBFeatures, + ProcessorFeatures.IVBTuning>; +def : ProcessorAlias<"core-avx-i", "ivybridge">; +def : ProcessorAlias<"core_3rd_gen_avx", "ivybridge">; + +def : ProcModel<"haswell", HaswellModel, ProcessorFeatures.HSWFeatures, + ProcessorFeatures.HSWTuning>; +def : ProcessorAlias<"core-avx2", "haswell">; +def : ProcessorAlias<"core_4th_gen_avx", "haswell">; +def : ProcessorAlias<"core_4th_gen_avx_tsx", "haswell">; + +def : ProcModel<"broadwell", BroadwellModel, ProcessorFeatures.BDWFeatures, + ProcessorFeatures.BDWTuning>; +def : ProcessorAlias<"core_5th_gen_avx", "broadwell">; +def : ProcessorAlias<"core_5th_gen_avx_tsx", "broadwell">; def : ProcModel<"skylake", SkylakeClientModel, ProcessorFeatures.SKLFeatures, ProcessorFeatures.SKLTuning>; // FIXME: define KNL scheduler model -foreach P = ["knl", "mic_avx512"] in { - def : ProcModel; -} +def : ProcModel<"knl", HaswellModel, ProcessorFeatures.KNLFeatures, + ProcessorFeatures.KNLTuning>; +def : ProcessorAlias<"mic_avx512", "knl">; def : ProcModel<"knm", HaswellModel, ProcessorFeatures.KNMFeatures, ProcessorFeatures.KNLTuning>; -foreach P = ["skylake-avx512", "skx", "skylake_avx512"] in { - def : ProcModel; -} +def : ProcModel<"skylake-avx512", SkylakeServerModel, ProcessorFeatures.SKXFeatures, + ProcessorFeatures.SKXTuning>; +def : ProcessorAlias<"skx", "skylake-avx512">; +def : ProcessorAlias<"skylake_avx512", "skylake-avx512">; def : ProcModel<"cascadelake", SkylakeServerModel, ProcessorFeatures.CLXFeatures, ProcessorFeatures.CLXTuning>; @@ -1895,16 +2099,14 @@ def : ProcModel<"cooperlake", SkylakeServerModel, ProcessorFeatures.CPXFeatures, ProcessorFeatures.CPXTuning>; def : ProcModel<"cannonlake", SkylakeServerModel, ProcessorFeatures.CNLFeatures, ProcessorFeatures.CNLTuning>; -foreach P = ["icelake-client", "icelake_client"] in { -def : ProcModel; -} +def : ProcessorAlias<"icelake_client", "icelake-client">; def : ProcModel<"rocketlake", IceLakeModel, ProcessorFeatures.ICLFeatures, ProcessorFeatures.ICLTuning>; -foreach P = ["icelake-server", "icelake_server"] in { -def : ProcModel; -} +def : ProcessorAlias<"icelake_server", "icelake-server">; def : ProcModel<"tigerlake", IceLakeModel, ProcessorFeatures.TGLFeatures, ProcessorFeatures.TGLTuning>; def : ProcModel<"sapphirerapids", SapphireRapidsModel, @@ -1914,28 +2116,25 @@ def : ProcModel<"alderlake", AlderlakePModel, // FIXME: Use Gracemont Schedule Model when it is ready. def : ProcModel<"gracemont", AlderlakePModel, ProcessorFeatures.ADLFeatures, ProcessorFeatures.GRTTuning>; -foreach P = ["sierraforest", "grandridge"] in { - def : ProcModel; -} +def : ProcessorAlias<"grandridge", "sierraforest">; def : ProcModel<"raptorlake", AlderlakePModel, ProcessorFeatures.ADLFeatures, ProcessorFeatures.ADLTuning>; def : ProcModel<"meteorlake", AlderlakePModel, ProcessorFeatures.ADLFeatures, ProcessorFeatures.ADLTuning>; def : ProcModel<"arrowlake", AlderlakePModel, ProcessorFeatures.ARLFeatures, ProcessorFeatures.ADLTuning>; -foreach P = ["arrowlake-s", "arrowlake_s"] in { -def : ProcModel; -} +def : ProcessorAlias<"arrowlake_s", "arrowlake-s">; def : ProcModel<"lunarlake", LunarlakePModel, ProcessorFeatures.ARLSFeatures, ProcessorFeatures.ADLTuning>; -foreach P = ["pantherlake", "wildcatlake"] in { -def : ProcModel; -} +def : ProcessorAlias<"wildcatlake", "pantherlake">; def : ProcModel<"novalake", AlderlakePModel, ProcessorFeatures.NVLFeatures, - ProcessorFeatures.ADLTuning>; + ProcessorFeatures.NVLTuning>; def : ProcModel<"clearwaterforest", AlderlakePModel, ProcessorFeatures.CWFFeatures, ProcessorFeatures.ADLTuning>; @@ -1943,12 +2142,11 @@ def : ProcModel<"emeraldrapids", SapphireRapidsModel, ProcessorFeatures.SPRFeatures, ProcessorFeatures.SPRTuning>; def : ProcModel<"graniterapids", SapphireRapidsModel, ProcessorFeatures.GNRFeatures, ProcessorFeatures.GNRTuning>; -foreach P = ["graniterapids-d", "graniterapids_d"] in { -def : ProcModel; -} +def : ProcessorAlias<"graniterapids_d", "graniterapids-d">; def : ProcModel<"diamondrapids", SapphireRapidsModel, - ProcessorFeatures.DMRFeatures, ProcessorFeatures.GNRTuning>; + ProcessorFeatures.DMRFeatures, ProcessorFeatures.DMRTuning>; // AMD CPUs. @@ -1959,37 +2157,38 @@ def : Proc<"k6-2", [FeatureX87, FeatureCX8, FeatureMMX, FeaturePRFCHW], def : Proc<"k6-3", [FeatureX87, FeatureCX8, FeatureMMX, FeaturePRFCHW], [TuningSlowUAMem16, TuningInsertVZEROUPPER]>; -foreach P = ["athlon", "athlon-tbird"] in { - def : Proc; -} - -foreach P = ["athlon-4", "athlon-xp", "athlon-mp"] in { - def : Proc; -} - -foreach P = ["k8", "opteron", "athlon64", "athlon-fx"] in { - def : Proc; -} - -foreach P = ["k8-sse3", "opteron-sse3", "athlon64-sse3"] in { - def : Proc; -} - -foreach P = ["amdfam10", "barcelona"] in { - def : Proc; -} +def : Proc<"athlon", [FeatureX87, FeatureCX8, FeatureCMOV, FeatureMMX, + FeaturePRFCHW, FeatureNOPL], + [TuningSlowSHLD, TuningSlowUAMem16, TuningInsertVZEROUPPER]>; +def : ProcessorAlias<"athlon-tbird", "athlon">; + +def : Proc<"athlon-4", [FeatureX87, FeatureCX8, FeatureCMOV, + FeatureSSE1, FeatureMMX, FeaturePRFCHW, FeatureFXSR, + FeatureNOPL], + [TuningSlowSHLD, TuningSlowUAMem16, TuningInsertVZEROUPPER]>; +def : ProcessorAlias<"athlon-xp", "athlon-4">; +def : ProcessorAlias<"athlon-mp", "athlon-4">; + +def : Proc<"k8", [FeatureX87, FeatureCX8, FeatureSSE2, FeatureMMX, FeaturePRFCHW, + FeatureFXSR, FeatureNOPL, FeatureX86_64, FeatureCMOV], + [TuningFastScalarShiftMasks, TuningSlowSHLD, TuningSlowUAMem16, + TuningSBBDepBreaking, TuningInsertVZEROUPPER]>; +def : ProcessorAlias<"opteron", "k8">; +def : ProcessorAlias<"athlon64", "k8">; +def : ProcessorAlias<"athlon-fx", "k8">; + +def : Proc<"k8-sse3", [FeatureX87, FeatureCX8, FeatureSSE3, FeatureMMX, + FeaturePRFCHW, FeatureFXSR, FeatureNOPL, FeatureCX16, + FeatureCMOV, FeatureX86_64], + [TuningFastScalarShiftMasks, TuningSlowSHLD, + TuningSlowUAMem16, TuningSBBDepBreaking, + TuningInsertVZEROUPPER]>; +def : ProcessorAlias<"opteron-sse3", "k8-sse3">; +def : ProcessorAlias<"athlon64-sse3", "k8-sse3">; + +def : Proc<"amdfam10", ProcessorFeatures.BarcelonaFeatures, + ProcessorFeatures.BarcelonaTuning>; +def : ProcessorAlias<"barcelona", "amdfam10">; // Bobcat def : Proc<"btver1", ProcessorFeatures.BtVer1Features, @@ -2026,6 +2225,16 @@ def : ProcModel<"znver5", Znver4Model, ProcessorFeatures.ZN5Features, def : ProcModel<"znver6", Znver4Model, ProcessorFeatures.ZN6Features, ProcessorFeatures.ZN6Tuning>; +// Hygon CPUs. + +def : ProcModel<"c86-4g-m4", C864GM4Model, ProcessorFeatures.C864GM4Features, + ProcessorFeatures.C864GM4Tuning>; +def : ProcessorAlias<"c86-4g-m6", "c86-4g-m4">; +def : ProcModel<"c86-4g-m7", C864GM7Model, ProcessorFeatures.C864GM7Features, + ProcessorFeatures.C864GM7Tuning>; +def : ProcModel<"c86-4g-m8", C864GM8Model, ProcessorFeatures.C864GM8Features, + ProcessorFeatures.C864GM8Tuning>; + def : Proc<"geode", [FeatureX87, FeatureCX8, FeatureMMX, FeaturePRFCHW], [TuningSlowUAMem16, TuningInsertVZEROUPPER]>; diff --git a/examples/llvm/Target/X86/X86RegisterInfo.td b/examples/llvm/Target/X86/X86RegisterInfo.td index 692e42a..63ad4b2 100644 --- a/examples/llvm/Target/X86/X86RegisterInfo.td +++ b/examples/llvm/Target/X86/X86RegisterInfo.td @@ -83,7 +83,7 @@ def R15B : X86Reg<"r15b", 15>; // To achieve this, we do the following things: // 1. Set CostPerUse=1 for registers that need prefix // 2. Consider callee-save register is never cheaper than a register w/ cost 1 -// 3. List caller-save register before callee-save regsiter in RegisterClass +// 3. List caller-save register before callee-save register in RegisterClass // or AllocationOrder // // NOTE: @@ -545,8 +545,8 @@ def SSP : X86Reg<"ssp", 0>; def GR8 : RegisterClass<"X86", [i8], 8, (add AL, CL, DL, AH, CH, DH, BL, BH, SIL, DIL, BPL, SPL, R8B, R9B, R10B, R11B, R16B, R17B, R18B, R19B, R22B, - R23B, R24B, R25B, R26B, R27B, R30B, R31B, R14B, - R15B, R12B, R13B, R20B, R21B, R28B, R29B)> { + R23B, R24B, R25B, R26B, R27B, R14B, R15B, R12B, + R13B, R20B, R21B, R28B, R29B, R30B, R31B)> { let AltOrders = [(sub GR8, AH, BH, CH, DH)]; let AltOrderSelect = [{ return MF.getSubtarget().is64Bit(); @@ -562,8 +562,8 @@ def GRH8 : RegisterClass<"X86", [i8], 8, def GR16 : RegisterClass<"X86", [i16], 16, (add AX, CX, DX, SI, DI, BX, BP, SP, R8W, R9W, R10W, R11W, R16W, R17W, R18W, R19W, R22W, R23W, R24W, - R25W, R26W, R27W, R30W, R31W, R14W, R15W, R12W, - R13W, R20W, R21W, R28W, R29W)>; + R25W, R26W, R27W, R14W, R15W, R12W, R13W, R20W, + R21W, R28W, R29W, R30W, R31W)>; let isAllocatable = 0 in def GRH16 : RegisterClass<"X86", [i16], 16, @@ -574,8 +574,8 @@ def GRH16 : RegisterClass<"X86", [i16], 16, def GR32 : RegisterClass<"X86", [i32], 32, (add EAX, ECX, EDX, ESI, EDI, EBX, EBP, ESP, R8D, R9D, R10D, R11D, R16D, R17D, R18D, R19D, R22D, R23D, - R24D, R25D, R26D, R27D, R30D, R31D, R14D, R15D, - R12D, R13D, R20D, R21D, R28D, R29D)>; + R24D, R25D, R26D, R27D, R14D, R15D, R12D, R13D, + R20D, R21D, R28D, R29D, R30D, R31D)>; // GR64 - 64-bit GPRs. This oddly includes RIP, which isn't accurate, since // RIP isn't really a register and it can't be used anywhere except in an @@ -584,8 +584,8 @@ def GR32 : RegisterClass<"X86", [i32], 32, // tests because of the inclusion of RIP in this register class. def GR64 : RegisterClass<"X86", [i64], 64, (add RAX, RCX, RDX, RSI, RDI, R8, R9, R10, R11, R16, R17, - R18, R19, R22, R23, R24, R25, R26, R27, R30, R31, RBX, - R14, R15, R12, R13, R20, R21, R28, R29, RBP, RSP, RIP)>; + R18, R19, R22, R23, R24, R25, R26, R27, RBX, R14, R15, + R12, R13, R20, R21, R28, R29, R30, R31, RBP, RSP, RIP)>; // GR64PLTSafe - 64-bit GPRs without R10, R11, RSP and RIP. Could be used when // emitting code for intrinsics, which use implict input registers. diff --git a/examples/mlir/Dialect/Affine/AffineOps.td b/examples/mlir/Dialect/Affine/AffineOps.td index b2a4cf7..cf2848f 100644 --- a/examples/mlir/Dialect/Affine/AffineOps.td +++ b/examples/mlir/Dialect/Affine/AffineOps.td @@ -15,14 +15,18 @@ include "mlir/Dialect/Arith/IR/ArithBase.td" include "mlir/Dialect/Affine/IR/AffineMemoryOpInterfaces.td" +include "mlir/Interfaces/AlignmentAttrInterface.td" include "mlir/Interfaces/ControlFlowInterfaces.td" +include "mlir/Interfaces/InferIntDivisibilityOpInterface.td" include "mlir/Interfaces/InferIntRangeInterface.td" include "mlir/Interfaces/InferTypeOpInterface.td" include "mlir/Interfaces/LoopLikeInterface.td" +include "mlir/Interfaces/MemorySlotInterfaces.td" include "mlir/Interfaces/SideEffectInterfaces.td" def Affine_Dialect : Dialect { let name = "affine"; + let useStrictPropertiesInAssemblyFormat = 1; let cppNamespace = "::mlir::affine"; let hasConstantMaterializer = 1; let dependentDialects = ["arith::ArithDialect", "ub::UBDialect"]; @@ -43,7 +47,9 @@ def ImplicitAffineTerminator : SingleBlockImplicitTerminator<"AffineYieldOp">; def AffineApplyOp : Affine_Op<"apply", - [Pure, DeclareOpInterfaceMethods]> { + [Pure, + DeclareOpInterfaceMethods, + DeclareOpInterfaceMethods]> { let summary = "affine apply operation"; let description = [{ The `affine.apply` operation applies an [affine mapping](#affine-maps) @@ -132,9 +138,10 @@ def AffineForOp : Affine_Op<"for", RecursiveMemoryEffects, DeclareOpInterfaceMethods, + "replaceWithAdditionalYields", "getStaticTripCount"]>, DeclareOpInterfaceMethods]> { + ["getEntrySuccessorOperands", "getSuccessorInputs"]>, + DeclareOpInterfaceMethods]> { let summary = "for operation"; let description = [{ Syntax: @@ -340,6 +347,7 @@ def AffineForOp : Affine_Op<"for", let hasCustomAssemblyFormat = 1; let hasFolder = 1; + let hasVerifier = 1; let hasRegionVerifier = 1; } @@ -476,11 +484,13 @@ class AffineLoadOpBase traits = []> : Affine_Op, DeclareOpInterfaceMethods, + DeclareOpInterfaceMethods, MemRefsNormalizable])> { let arguments = (ins Arg:$memref, Variadic:$indices, - AffineMapAttr:$map); + AffineMapAttr:$map, + OptionalAttr>:$alignment); code extraClassDeclarationBase = [{ /// Returns the operand index of the memref. @@ -524,18 +534,27 @@ def AffineLoadOp : AffineLoadOpBase<"load"> { ```mlir %1 = affine.load %0[%i0 + symbol(%n), %i1 + symbol(%m)] : memref<100x100xf32> ``` + + An optional `alignment` attribute allows to specify the byte alignment of + the load operation. It must be a positive power of 2. The operation must + access memory at an address aligned to this boundary. Violations may lead + to architecture-specific faults or performance penalties. + A value of 0 indicates no specific alignment requirement. }]; let results = (outs AnyType:$result); let builders = [ /// Builds an affine load op with the specified map and operands. - OpBuilder<(ins "AffineMap":$map, "ValueRange":$operands)>, + OpBuilder<(ins "AffineMap":$map, "ValueRange":$operands, + CArg<"llvm::MaybeAlign", "llvm::MaybeAlign()">:$alignment)>, /// Builds an affine load op with an identity map and operands. - OpBuilder<(ins "Value":$memref, CArg<"ValueRange", "{}">:$indices)>, + OpBuilder<(ins "Value":$memref, CArg<"ValueRange", "{}">:$indices, + CArg<"llvm::MaybeAlign", "llvm::MaybeAlign()">:$alignment)>, /// Builds an affine load op with the specified map and its operands. OpBuilder<(ins "Value":$memref, "AffineMap":$map, - "ValueRange":$mapOperands)> + "ValueRange":$mapOperands, + CArg<"llvm::MaybeAlign", "llvm::MaybeAlign()">:$alignment)> ]; let extraClassDeclaration = extraClassDeclarationBase; @@ -570,7 +589,8 @@ class AffineMinMaxOpBase traits = []> : let hasVerifier = 1; } -def AffineMinOp : AffineMinMaxOpBase<"min", [Pure]> { +def AffineMinOp : AffineMinMaxOpBase<"min", + [Pure, DeclareOpInterfaceMethods]> { let summary = "min operation"; let description = [{ Syntax: @@ -594,7 +614,8 @@ def AffineMinOp : AffineMinMaxOpBase<"min", [Pure]> { }]; } -def AffineMaxOp : AffineMinMaxOpBase<"max", [Pure]> { +def AffineMaxOp : AffineMinMaxOpBase<"max", + [Pure, DeclareOpInterfaceMethods]> { let summary = "max operation"; let description = [{ The `affine.max` operation computes the maximum value result from a multi-result @@ -841,6 +862,7 @@ class AffineStoreOpBase traits = []> : Affine_Op, DeclareOpInterfaceMethods, + DeclareOpInterfaceMethods, MemRefsNormalizable])> { code extraClassDeclarationBase = [{ /// Returns the operand index of the value to be stored. @@ -887,19 +909,28 @@ def AffineStoreOp : AffineStoreOpBase<"store"> { ```mlir affine.store %v0, %0[%i0 + symbol(%n), %i1 + symbol(%m)] : memref<100x100xf32> ``` + + An optional `alignment` attribute allows to specify the byte alignment of + the store operation. It must be a positive power of 2. The operation must + access memory at an address aligned to this boundary. Violations may lead + to architecture-specific faults or performance penalties. + A value of 0 indicates no specific alignment requirement. }]; let arguments = (ins AnyType:$value, Arg:$memref, Variadic:$indices, - AffineMapAttr:$map); + AffineMapAttr:$map, + OptionalAttr>:$alignment); let skipDefaultBuilders = 1; let builders = [ OpBuilder<(ins "Value":$valueToStore, "Value":$memref, - "ValueRange":$indices)>, + "ValueRange":$indices, + CArg<"llvm::MaybeAlign", "llvm::MaybeAlign()">:$alignment)>, OpBuilder<(ins "Value":$valueToStore, "Value":$memref, "AffineMap":$map, - "ValueRange":$mapOperands)> + "ValueRange":$mapOperands, + CArg<"llvm::MaybeAlign", "llvm::MaybeAlign()">:$alignment)> ]; let extraClassDeclaration = extraClassDeclarationBase; @@ -978,13 +1009,16 @@ def AffineVectorLoadOp : AffineLoadOpBase<"vector_load"> { let builders = [ /// Builds an affine vector load op with the specified map and operands. OpBuilder<(ins "VectorType":$resultType, "AffineMap":$map, - "ValueRange":$operands)>, + "ValueRange":$operands, + CArg<"llvm::MaybeAlign", "llvm::MaybeAlign()">:$alignment)>, /// Builds an affine vector load op with an identity map and operands. OpBuilder<(ins "VectorType":$resultType, "Value":$memref, - CArg<"ValueRange", "{}">:$indices)>, + CArg<"ValueRange", "{}">:$indices, + CArg<"llvm::MaybeAlign", "llvm::MaybeAlign()">:$alignment)>, /// Builds an affine vector load op with the specified map and its operands. OpBuilder<(ins "VectorType":$resultType, "Value":$memref, - "AffineMap":$map, "ValueRange":$mapOperands)> + "AffineMap":$map, "ValueRange":$mapOperands, + CArg<"llvm::MaybeAlign", "llvm::MaybeAlign()">:$alignment)> ]; let extraClassDeclaration = extraClassDeclarationBase # [{ @@ -1044,14 +1078,17 @@ def AffineVectorStoreOp : AffineStoreOpBase<"vector_store"> { Arg:$memref, Variadic:$indices, - AffineMapAttr:$map); + AffineMapAttr:$map, + OptionalAttr>:$alignment); let skipDefaultBuilders = 1; let builders = [ OpBuilder<(ins "Value":$valueToStore, "Value":$memref, - "ValueRange":$indices)>, + "ValueRange":$indices, + CArg<"llvm::MaybeAlign", "llvm::MaybeAlign()">:$alignment)>, OpBuilder<(ins "Value":$valueToStore, "Value":$memref, "AffineMap":$map, - "ValueRange":$mapOperands)> + "ValueRange":$mapOperands, + CArg<"llvm::MaybeAlign", "llvm::MaybeAlign()">:$alignment)> ]; let extraClassDeclaration = extraClassDeclarationBase # [{ @@ -1071,6 +1108,7 @@ def AffineVectorStoreOp : AffineStoreOpBase<"vector_store"> { def AffineDelinearizeIndexOp : Affine_Op<"delinearize_index", [Pure, Elementwise, + DeclareOpInterfaceMethods, // Infer linear_index type from the first result type during parsing. TypesMatchWith<"linear_index type must match result types", "multi_index", "linear_index", "$_self[0]"> diff --git a/examples/mlir/Dialect/Arith/ArithOps.td b/examples/mlir/Dialect/Arith/ArithOps.td index f359070..39d6838 100644 --- a/examples/mlir/Dialect/Arith/ArithOps.td +++ b/examples/mlir/Dialect/Arith/ArithOps.td @@ -13,6 +13,7 @@ include "mlir/Dialect/Arith/IR/ArithBase.td" include "mlir/Dialect/Arith/IR/ArithOpsInterfaces.td" include "mlir/Interfaces/CastInterfaces.td" include "mlir/Interfaces/ControlFlowInterfaces.td" +include "mlir/Interfaces/InferIntDivisibilityOpInterface.td" include "mlir/Interfaces/InferIntRangeInterface.td" include "mlir/Interfaces/InferTypeOpInterface.td" include "mlir/Interfaces/SideEffectInterfaces.td" @@ -223,6 +224,7 @@ def Arith_ConstantOp : Op, AllTypesMatch<["value", "result"]>, + DeclareOpInterfaceMethods, DeclareOpInterfaceMethods]> { let summary = "integer or floating point constant"; let description = [{ @@ -242,12 +244,8 @@ def Arith_ConstantOp : Op { +def Arith_AddIOp : Arith_IntBinaryOpWithOverflowFlags<"addi", + [Commutative, DeclareOpInterfaceMethods]> { let summary = "integer addition operation"; let description = [{ Performs N-bit addition on the operands. The operands are interpreted as @@ -359,11 +358,65 @@ def Arith_AddUIExtendedOp : Arith_Op<"addui_extended", [Pure, Commutative, }]; } +//===----------------------------------------------------------------------===// +// SubUIExtendedOp +//===----------------------------------------------------------------------===// + +def Arith_SubUIExtendedOp : Arith_Op<"subui_extended", [Pure, + AllTypesMatch<["lhs", "rhs", "diff"]>]> { + let summary = [{ + extended unsigned integer subtraction operation returning difference and + borrow bit + }]; + + let description = [{ + Performs (N+1)-bit subtraction on zero-extended operands. Returns two + results: the N-bit difference (same type as both operands), and the borrow + bit (boolean-like), where `1` indicates unsigned subtraction underflow + (i.e. `lhs < rhs` when interpreted as unsigned), while `0` indicates no + underflow. + + Example: + + ```mlir + // Scalar subtraction. + %diff, %borrow = arith.subui_extended %b, %c : i64, i1 + + // Vector element-wise subtraction. + %d:2 = arith.subui_extended %e, %f : vector<4xi32>, vector<4xi1> + + // Tensor element-wise subtraction. + %x:2 = arith.subui_extended %y, %z : tensor<4x?xi8>, tensor<4x?xi1> + ``` + }]; + + let arguments = (ins Arith_SignlessIntegerOrIndexLike:$lhs, Arith_SignlessIntegerOrIndexLike:$rhs); + let results = (outs Arith_SignlessIntegerOrIndexLike:$diff, BoolLike:$borrow); + let assemblyFormat = [{ + $lhs `,` $rhs attr-dict `:` type($diff) `,` type($borrow) + }]; + + let builders = [ + OpBuilder<(ins "Value":$lhs, "Value":$rhs), [{ + build($_builder, $_state, lhs.getType(), ::getI1SameShape(lhs.getType()), + lhs, rhs); + }]> + ]; + + let hasFolder = 1; + let hasCanonicalizer = 1; + + let extraClassDeclaration = [{ + std::optional> getShapeForUnroll(); + }]; +} + //===----------------------------------------------------------------------===// // SubIOp //===----------------------------------------------------------------------===// -def Arith_SubIOp : Arith_IntBinaryOpWithOverflowFlags<"subi"> { +def Arith_SubIOp : Arith_IntBinaryOpWithOverflowFlags<"subi", + [DeclareOpInterfaceMethods]> { let summary = [{ Integer subtraction operation. }]; @@ -408,7 +461,9 @@ def Arith_SubIOp : Arith_IntBinaryOpWithOverflowFlags<"subi"> { //===----------------------------------------------------------------------===// def Arith_MulIOp : Arith_IntBinaryOpWithOverflowFlags<"muli", - [Commutative, DeclareOpInterfaceMethods] + [Commutative, + DeclareOpInterfaceMethods, + DeclareOpInterfaceMethods] > { let summary = [{ Integer multiplication operation. @@ -540,7 +595,8 @@ def Arith_MulUIExtendedOp : Arith_Op<"mului_extended", [Pure, Commutative, //===----------------------------------------------------------------------===// def Arith_DivUIOp : Arith_IntBinaryOpWithExactFlag<"divui", - [ConditionallySpeculatable]> { + [ConditionallySpeculatable, + DeclareOpInterfaceMethods]> { let summary = "unsigned integer division operation"; let description = [{ Unsigned integer division. Rounds towards zero. Treats the leading bit as @@ -1057,6 +1113,7 @@ def Arith_AddFOp : Arith_FloatBinaryOpWithRoundingMode<"addf", [Commutative]> { %a = arith.addf %b, %c to_nearest_even : f64 ``` }]; + let hasCanonicalizer = 1; let hasFolder = 1; } @@ -1138,7 +1195,8 @@ def Arith_MaxNumFOp : Arith_FloatBinaryOp<"maxnumf", [Commutative]> { // MaxSIOp //===----------------------------------------------------------------------===// -def Arith_MaxSIOp : Arith_TotalIntBinaryOp<"maxsi", [Commutative]> { +def Arith_MaxSIOp : Arith_TotalIntBinaryOp<"maxsi", + [Commutative, DeclareOpInterfaceMethods]> { let summary = "signed integer maximum operation"; let hasFolder = 1; } @@ -1147,7 +1205,8 @@ def Arith_MaxSIOp : Arith_TotalIntBinaryOp<"maxsi", [Commutative]> { // MaxUIOp //===----------------------------------------------------------------------===// -def Arith_MaxUIOp : Arith_TotalIntBinaryOp<"maxui", [Commutative]> { +def Arith_MaxUIOp : Arith_TotalIntBinaryOp<"maxui", + [Commutative, DeclareOpInterfaceMethods]> { let summary = "unsigned integer maximum operation"; let hasFolder = 1; } @@ -1197,7 +1256,8 @@ def Arith_MinNumFOp : Arith_FloatBinaryOp<"minnumf", [Commutative]> { // MinSIOp //===----------------------------------------------------------------------===// -def Arith_MinSIOp : Arith_TotalIntBinaryOp<"minsi", [Commutative]> { +def Arith_MinSIOp : Arith_TotalIntBinaryOp<"minsi", + [Commutative, DeclareOpInterfaceMethods]> { let summary = "signed integer minimum operation"; let hasFolder = 1; } @@ -1206,7 +1266,8 @@ def Arith_MinSIOp : Arith_TotalIntBinaryOp<"minsi", [Commutative]> { // MinUIOp //===----------------------------------------------------------------------===// -def Arith_MinUIOp : Arith_TotalIntBinaryOp<"minui", [Commutative]> { +def Arith_MinUIOp : Arith_TotalIntBinaryOp<"minui", + [Commutative, DeclareOpInterfaceMethods]> { let summary = "unsigned integer minimum operation"; let hasFolder = 1; } @@ -1439,6 +1500,7 @@ def Arith_ScalingExtFOp %h = arith.scaling_extf %i, %f : vector<32xf4E2M1FN>, vector<32xf8E8M0FNU> to vector<32xbf16> ``` }]; + let hasFolder = 1; let hasVerifier = 1; let assemblyFormat = [{ $in `,` $scale (`fastmath` `` $fastmath^)? attr-dict `:` @@ -1628,6 +1690,7 @@ def Arith_ScalingTruncFOp %h = arith.scaling_truncf %i, %f : vector<32xbf16>, vector<32xf8E8M0FNU> to vector<32xf4E2M1FN> ``` }]; + let hasFolder = 1; let hasVerifier = 1; let assemblyFormat = [{ $in `,` $scale ($roundingmode^)? (`fastmath` `` $fastmath^)? attr-dict `:` @@ -1951,6 +2014,7 @@ class BooleanConditionOrMatchingShape : def SelectOp : Arith_Op<"select", [Pure, AllTypesMatch<["true_value", "false_value", "result"]>, BooleanConditionOrMatchingShape<"condition", "result">, + DeclareOpInterfaceMethods, DeclareOpInterfaceMethods, DeclareOpInterfaceMethods]> { let summary = "select operation"; diff --git a/examples/mlir/Dialect/Async/AsyncDialect.td b/examples/mlir/Dialect/Async/AsyncDialect.td index eb1d76a..f2c328a 100644 --- a/examples/mlir/Dialect/Async/AsyncDialect.td +++ b/examples/mlir/Dialect/Async/AsyncDialect.td @@ -21,6 +21,7 @@ include "mlir/IR/OpBase.td" def AsyncDialect : Dialect { let name = "async"; + let useStrictPropertiesInAssemblyFormat = 1; let cppNamespace = "::mlir::async"; let summary = "Types and operations for async dialect"; diff --git a/examples/mlir/Dialect/Async/AsyncOps.td b/examples/mlir/Dialect/Async/AsyncOps.td index 2cebeac..cc21bc1 100644 --- a/examples/mlir/Dialect/Async/AsyncOps.td +++ b/examples/mlir/Dialect/Async/AsyncOps.td @@ -107,7 +107,8 @@ def Async_ExecuteOp : } def Async_FuncOp : Async_Op<"func", - [FunctionOpInterface, IsolatedFromAbove, OpAsmOpInterface]> { + [SymbolName, SymbolVisibility, FunctionOpInterface, IsolatedFromAbove, + OpAsmOpInterface]> { let summary = "async function operation"; let description = [{ An async function is like a normal function, but supports non-blocking @@ -174,7 +175,9 @@ def Async_FuncOp : Async_Op<"func", unsigned getNumResults() {return getResultTypes().size();} /// Is the async func stateful - bool isStateful() { return isa(getFunctionType().getResult(0));} + bool isStateful() { + return isa(getFunctionType().getResult(0)); + } //===------------------------------------------------------------------===// // OpAsmOpInterface Methods @@ -259,17 +262,20 @@ def Async_CallOp : Async_Op<"call", /// Return the callee of this operation. CallInterfaceCallable getCallableForCallee() { - return (*this)->getAttrOfType("callee"); + return getCalleeAttr(); } /// Set the callee for this operation. void setCalleeFromCallable(CallInterfaceCallable callee) { - (*this)->setAttr("callee", cast(callee)); + setCalleeAttr(cast(cast(callee))); } }]; let assemblyFormat = [{ - $callee `(` $operands `)` attr-dict `:` functional-type($operands, results) + $callee `(` $operands `)` + (`arg_attrs` `=` $arg_attrs^)? + (`res_attrs` `=` $res_attrs^)? + attr-dict `:` functional-type($operands, results) }]; } @@ -484,7 +490,7 @@ def Async_CoroEndOp : Async_Op<"coro.end"> { let description = [{ The `async.coro.end` marks the point where a coroutine needs to return control back to the caller if it is not an initial invocation of the - coroutine. It the start part of the coroutine is is no-op. + coroutine. In the start part of the coroutine it is no-op. }]; let arguments = (ins Async_CoroHandleType:$handle); @@ -703,7 +709,7 @@ def Async_RuntimeAddRefOp : Async_Op<"runtime.add_ref"> { ConfinedAttr:$count); let assemblyFormat = [{ - $operand attr-dict `:` type($operand) + $operand `count` `=` $count attr-dict `:` type($operand) }]; } @@ -718,7 +724,7 @@ def Async_RuntimeDropRefOp : Async_Op<"runtime.drop_ref"> { ConfinedAttr:$count); let assemblyFormat = [{ - $operand attr-dict `:` type($operand) + $operand `count` `=` $count attr-dict `:` type($operand) }]; } diff --git a/examples/mlir/Dialect/Bufferization/BufferizationEnums.td b/examples/mlir/Dialect/Bufferization/BufferizationEnums.td index bafa846..a8cad26 100644 --- a/examples/mlir/Dialect/Bufferization/BufferizationEnums.td +++ b/examples/mlir/Dialect/Bufferization/BufferizationEnums.td @@ -24,4 +24,12 @@ def LayoutMapOption : I32EnumAttr<"LayoutMapOption", let cppNamespace = "::mlir::bufferization"; } +def MemoryPlannerAlgorithm : I32EnumAttr<"MemoryPlannerAlgorithm", + "memory planning algorithm", [ + I32EnumAttrCase<"Trivial", 0, "trivial">, + I32EnumAttrCase<"BestFit", 1, "best-fit"> +]> { + let cppNamespace = "::mlir::bufferization"; +} + #endif // BUFFERIZATION_ENUMS diff --git a/examples/mlir/Dialect/Bufferization/BufferizationOps.td b/examples/mlir/Dialect/Bufferization/BufferizationOps.td index a9b2b9f..30317c2 100644 --- a/examples/mlir/Dialect/Bufferization/BufferizationOps.td +++ b/examples/mlir/Dialect/Bufferization/BufferizationOps.td @@ -11,7 +11,6 @@ include "mlir/Dialect/Bufferization/IR/AllocationOpInterface.td" include "mlir/Dialect/Bufferization/IR/BufferViewFlowOpInterface.td" -include "mlir/Dialect/Bufferization/IR/BufferizableOpInterface.td" include "mlir/Dialect/Bufferization/IR/BufferizationTypeInterfaces.td" include "mlir/Dialect/Bufferization/IR/BufferizationBase.td" include "mlir/Interfaces/DestinationStyleOpInterface.td" @@ -27,7 +26,7 @@ class Bufferization_Op traits = []> //===----------------------------------------------------------------------===// def Bufferization_AllocTensorOp : Bufferization_Op<"alloc_tensor", - [AttrSizedOperandSegments, BufferizableOpInterface, + [AttrSizedOperandSegments, DeclareOpInterfaceMethods]> { let summary = "allocate buffer for a tensor"; @@ -93,29 +92,6 @@ def Bufferization_AllocTensorOp : Bufferization_Op<"alloc_tensor", let results = (outs AnyTensor:$result); let extraClassDeclaration = [{ - LogicalResult bufferize(RewriterBase &rewriter, - const BufferizationOptions &options, - BufferizationState &state); - - bool resultBufferizesToMemoryWrite(OpResult opResult, - const AnalysisState &state); - - bool bufferizesToAllocation(Value value) { return true; } - - bool bufferizesToMemoryRead(OpOperand &opOperand, - const AnalysisState &state); - - bool bufferizesToMemoryWrite(OpOperand &opOperand, - const AnalysisState &state); - - AliasingValueList getAliasingValues( - OpOperand &opOperand, const AnalysisState &state); - - FailureOr getBufferType( - Value value, const BufferizationOptions &options, - const BufferizationState &state, - SmallVector &invocationStack); - RankedTensorType getType() { return ::llvm::cast(getResult().getType()); } @@ -219,7 +195,7 @@ def Bufferization_CloneOp : Bufferization_Op<"clone", [ def Bufferization_MaterializeInDestinationOp : Bufferization_Op<"materialize_in_destination", [AllElementTypesMatch<["source", "dest"]>, - BufferizableOpInterface, DestinationStyleOpInterface, + DestinationStyleOpInterface, DeclareOpInterfaceMethods, DeclareOpInterfaceMethods:$result); let extraClassDeclaration = [{ - LogicalResult bufferize(RewriterBase &rewriter, - const BufferizationOptions &options, - BufferizationState &state); - - bool bufferizesToMemoryRead(OpOperand &opOperand, - const AnalysisState &state); - - bool bufferizesToMemoryWrite(OpOperand &opOperand, - const AnalysisState &state); - - bool bufferizesToElementwiseAccess(const AnalysisState &state, - ArrayRef opOperands); - - bool mustBufferizeInPlace(OpOperand &opOperand, - const AnalysisState &state); - - AliasingValueList getAliasingValues( - OpOperand &opOperand, const AnalysisState &state); - RankedTensorType getType() { return ::llvm::cast(getResult().getType()); } MutableOperandRange getDpsInitsMutable(); - - bool isWritable(Value value, const AnalysisState &state); }]; let builders = [ @@ -329,8 +284,7 @@ def Bufferization_MaterializeInDestinationOp // DeallocTensorOp //===----------------------------------------------------------------------===// -def Bufferization_DeallocTensorOp : Bufferization_Op<"dealloc_tensor", - [BufferizableOpInterface]> { +def Bufferization_DeallocTensorOp : Bufferization_Op<"dealloc_tensor"> { string summary = "release underlying storage format of given tensor"; string description = [{ `bufferization.dealloc_tensor` is a buffer deallocation in tensor land. This @@ -360,27 +314,6 @@ def Bufferization_DeallocTensorOp : Bufferization_Op<"dealloc_tensor", let arguments = (ins AnyTensor:$tensor); let results = (outs); let assemblyFormat = "$tensor attr-dict `:` type($tensor)"; - - let extraClassDeclaration = [{ - bool bufferizesToMemoryRead(OpOperand &opOperand, - const AnalysisState &state) const { - return false; - } - - bool bufferizesToMemoryWrite(OpOperand &opOperand, - const AnalysisState &state) const { - return false; - } - - AliasingValueList getAliasingValues( - OpOperand &opOperand, const AnalysisState &state) const { - return {}; - } - - LogicalResult bufferize(RewriterBase &rewriter, - const BufferizationOptions &options, - BufferizationState &state); - }]; } //===----------------------------------------------------------------------===// @@ -396,7 +329,6 @@ class Bufferization_TensorAndBufferMatch : PredOpT >; def Bufferization_ToTensorOp : Bufferization_Op<"to_tensor", [ - BufferizableOpInterface, SameOperandsAndResultShape, SameOperandsAndResultElementType, Bufferization_TensorAndBufferMatch<"result", "buffer"> @@ -464,25 +396,6 @@ def Bufferization_ToTensorOp : Bufferization_Op<"to_tensor", [ ::mlir::bufferization::TensorLikeType getType() { return getResult().getType(); } - - //===------------------------------------------------------------------===// - // BufferizableOpInterface implementation - //===------------------------------------------------------------------===// - - LogicalResult bufferize(RewriterBase &rewriter, - const BufferizationOptions &options, - BufferizationState &state) const { - // to_tensor/to_buffer pairs fold away after bufferization. - return success(); - } - - bool isWritable(Value value, const AnalysisState &state); - - FailureOr getBufferType( - Value value, const BufferizationOptions &options, - const BufferizationState &state, SmallVector &invocationStack) { - return getBuffer().getType(); - } }]; let assemblyFormat = [{ @@ -500,7 +413,6 @@ def Bufferization_ToTensorOp : Bufferization_Op<"to_tensor", [ //===----------------------------------------------------------------------===// def Bufferization_ToBufferOp : Bufferization_Op<"to_buffer", [ - BufferizableOpInterface, SameOperandsAndResultShape, SameOperandsAndResultElementType, Pure, @@ -527,38 +439,6 @@ def Bufferization_ToBufferOp : Bufferization_Op<"to_buffer", [ let arguments = (ins Bufferization_TensorLikeTypeInterface:$tensor, UnitAttr:$read_only); let results = (outs Bufferization_BufferLikeTypeInterface:$buffer); - let extraClassDeclaration = [{ - //===------------------------------------------------------------------===// - // BufferizableOpInterface implementation - //===------------------------------------------------------------------===// - - // Note: ToBufferOp / ToTensorOp are temporary ops that are inserted at the - // bufferization boundary. When One-Shot bufferization is complete, there - // should be no such ops left over. If `allowUnknownOps` (or after running a - // partial bufferization pass), such ops may be part of the resulting IR, - // but such IR may no longer be analyzable by One-Shot analysis. - - bool bufferizesToMemoryRead(OpOperand &opOperand, - const AnalysisState &state) const { - // It is unknown whether the resulting memref will be read or not. - return true; - } - - bool bufferizesToMemoryWrite(OpOperand &opOperand, - const AnalysisState &state) { - return !getReadOnly(); - } - - AliasingValueList getAliasingValues( - OpOperand &opOperand, const AnalysisState &state) const { - return {}; - } - - LogicalResult bufferize(RewriterBase &rewriter, - const BufferizationOptions &options, - BufferizationState &state); - }]; - let assemblyFormat = [{ $tensor (`read_only` $read_only^)? attr-dict `:` type($tensor) `to` type($buffer) }]; diff --git a/examples/mlir/Dialect/Bufferization/BufferizationTypeInterfaces.td b/examples/mlir/Dialect/Bufferization/BufferizationTypeInterfaces.td index fb6fc4f..6bbc6cb 100644 --- a/examples/mlir/Dialect/Bufferization/BufferizationTypeInterfaces.td +++ b/examples/mlir/Dialect/Bufferization/BufferizationTypeInterfaces.td @@ -24,19 +24,9 @@ def Bufferization_TensorLikeTypeInterface }]; let methods = [ - InterfaceMethod<[{ - Returns a BufferLike type for this TensorLike type. - }], - /*retTy=*/"::mlir::FailureOr<::mlir::bufferization::BufferLikeType>", - /*methodName=*/"getBufferType", - /*args=*/(ins - "const ::mlir::bufferization::BufferizationOptions &":$options, - "::llvm::function_ref<::mlir::InFlightDiagnostic()>":$emitError - ) - >, InterfaceMethod<[{ Returns whether a BufferLike type is compatible to this TensorLike type. - The BufferLike type is assumed to be created by getBufferType(). + The BufferLike type is assumed to be created by unknown type converter. }], /*retTy=*/"::mlir::LogicalResult", /*methodName=*/"verifyCompatibleBufferType", diff --git a/examples/mlir/Dialect/ControlFlow/ControlFlowOps.td b/examples/mlir/Dialect/ControlFlow/ControlFlowOps.td index a441fd8..0e4c4eb 100644 --- a/examples/mlir/Dialect/ControlFlow/ControlFlowOps.td +++ b/examples/mlir/Dialect/ControlFlow/ControlFlowOps.td @@ -23,6 +23,7 @@ def ControlFlow_Dialect : Dialect { let name = "cf"; let cppNamespace = "::mlir::cf"; let dependentDialects = ["arith::ArithDialect"]; + let useStrictPropertiesInAssemblyFormat = 1; let description = [{ This dialect contains low-level, i.e. non-region based, control flow constructs. These constructs generally represent control flow directly diff --git a/examples/mlir/Dialect/Func/FuncOps.td b/examples/mlir/Dialect/Func/FuncOps.td index a31b860..d7a0d5f 100644 --- a/examples/mlir/Dialect/Func/FuncOps.td +++ b/examples/mlir/Dialect/Func/FuncOps.td @@ -22,6 +22,7 @@ def Func_Dialect : Dialect { let name = "func"; let cppNamespace = "::mlir::func"; let hasConstantMaterializer = 1; + let useStrictPropertiesInAssemblyFormat = 1; } // Base class for Func dialect ops. @@ -110,17 +111,18 @@ def CallOp : Func_Op<"call", /// Return the callee of this operation. CallInterfaceCallable getCallableForCallee() { - return (*this)->getAttrOfType("callee"); + return getCalleeAttr(); } /// Set the callee for this operation. void setCalleeFromCallable(CallInterfaceCallable callee) { - (*this)->setAttr("callee", cast(callee)); + setCalleeAttr(cast(cast(callee))); } }]; let assemblyFormat = [{ - $callee `(` $operands `)` attr-dict `:` functional-type($operands, results) + $callee `(` $operands `)` prop-dict attr-dict `:` + functional-type($operands, results) }]; } @@ -197,7 +199,12 @@ def CallIndirectOp : Func_Op<"call_indirect", [ let hasCanonicalizeMethod = 1; let assemblyFormat = [{ - $callee `(` $callee_operands `)` attr-dict `:` type($callee) + $callee `(` $callee_operands `)` + oilist<`,`>( + `arg_attrs` `=` $arg_attrs + | `res_attrs` `=` $res_attrs + ) + attr-dict `:` type($callee) }]; } @@ -249,7 +256,8 @@ def ConstantOp : Func_Op<"constant", def FuncOp : Func_Op<"func", [ AffineScope, AutomaticAllocationScope, - FunctionOpInterface, IsolatedFromAbove, OpAsmOpInterface + SymbolName, SymbolVisibility, FunctionOpInterface, IsolatedFromAbove, + OpAsmOpInterface ]> { let summary = "An operation with a name containing a single `SSACFG` region"; let description = [{ diff --git a/examples/mlir/Dialect/GPU/GPUOps.td b/examples/mlir/Dialect/GPU/GPUOps.td index a552558..2aa5219 100644 --- a/examples/mlir/Dialect/GPU/GPUOps.td +++ b/examples/mlir/Dialect/GPU/GPUOps.td @@ -45,7 +45,8 @@ class GPU_IndexOp traits = []> : DeclareOpInterfaceMethods])>, Arguments<(ins GPU_DimensionAttr:$dimension, OptionalAttr:$upper_bound)>, Results<(outs Index)> { - let assemblyFormat = "$dimension (`upper_bound` $upper_bound^)? attr-dict"; + let assemblyFormat = + "enum($dimension) (`upper_bound` $upper_bound^)? attr-dict"; let extraClassDefinition = [{ void $cppClass::getAsmResultNames( llvm::function_ref setNameFn) { @@ -346,7 +347,8 @@ def GPU_OptionalDimSizeHintAttr : ConfinedAttr, >; def GPU_GPUFuncOp : GPU_Op<"func", [ - HasParent<"GPUModuleOp">, AutomaticAllocationScope, FunctionOpInterface, + HasParent<"GPUModuleOp">, AutomaticAllocationScope, SymbolName, + SymbolVisibility, FunctionOpInterface, IsolatedFromAbove, AffineScope ]> { let summary = "Function executable on a GPU"; @@ -419,7 +421,9 @@ def GPU_GPUFuncOp : GPU_Op<"func", [ attribution. }]; - let arguments = (ins TypeAttrOf:$function_type, + let arguments = (ins SymbolNameAttr:$sym_name, + OptionalAttr:$sym_visibility, + TypeAttrOf:$function_type, OptionalAttr:$arg_attrs, OptionalAttr:$res_attrs, OptionalAttr:$workgroup_attrib_attrs, @@ -447,7 +451,7 @@ def GPU_GPUFuncOp : GPU_Op<"func", [ bool isKernel() { if (getKernel()) return true; - return (*this)->getAttrOfType( + return (*this)->getDiscardableAttrOfType( GPUDialect::getKernelFuncAttrName()) != nullptr; } @@ -637,13 +641,21 @@ def GPU_LaunchFuncOp :GPU_Op<"launch_func", [ operation has a symbol attribute named `kernel` to identify the fully specified kernel function to launch (both the gpu.module and func). - The `gpu.launch_func` supports async dependencies: the kernel does not start + By default, the host implicitly blocks until kernel execution has completed. + + Otherwise, the operation supports two async models. + + The first one is dependency-based and is enabled when the `async` keyword is + present in text form, and corresponds to when the operation produces the + optional token result of type `!gpu.async.token`. Other async GPU ops can + take this token as dependency. In this case, the `gpu.launch_func` does not + block, and supports specifying async dependencies: the kernel does not start executing until the ops producing those async dependencies have completed. - By the default, the host implicitly blocks until kernel execution has - completed. If the `async` keyword is present, the host does not block but - instead a `!gpu.async.token` is returned. Other async GPU ops can take this - token as dependency. + The second async model is stream-based. When `asyncObject` is present, the + launch operation is queued to execute on the queue represented by it. + + The two async models are mutually exclusive. The operation requires at least the grid and block sizes along the x,y,z dimensions as arguments. When a lower-dimensional kernel is required, @@ -734,16 +746,13 @@ def GPU_LaunchFuncOp :GPU_Op<"launch_func", [ "ValueRange":$kernelOperands, CArg<"Type", "nullptr">:$asyncTokenType, CArg<"ValueRange", "{}">:$asyncDependencies, + CArg<"Value", "nullptr">:$asyncObject, CArg<"std::optional", "std::nullopt">:$clusterSize)>, OpBuilder<(ins "SymbolRefAttr":$kernel, "KernelDim3":$gridSize, "KernelDim3":$blockSize, "Value":$dynamicSharedMemorySize, "ValueRange":$kernelOperands, - "Type":$asyncTokenType, + CArg<"Type", "nullptr">:$asyncTokenType, CArg<"ValueRange", "{}">:$asyncDependencies, - CArg<"std::optional", "std::nullopt">:$clusterSize)>, - OpBuilder<(ins "SymbolRefAttr":$kernel, "KernelDim3":$gridSize, - "KernelDim3":$blockSize, "Value":$dynamicSharedMemorySize, - "ValueRange":$kernelOperands, CArg<"Value", "nullptr">:$asyncObject, CArg<"std::optional", "std::nullopt">:$clusterSize)> ]; @@ -813,20 +822,17 @@ def GPU_LaunchOp : GPU_Op<"launch", [ UnitAttr:$cooperative, OptionalAttr:$module, OptionalAttr:$function, - OptionalAttr>:$workgroup_attributions)>, + OptionalAttr>:$workgroup_attributions, + Optional:$asyncObject)>, Results<(outs Optional:$asyncToken)> { let summary = "GPU kernel launch operation"; let description = [{ Launch a kernel on the specified grid of thread blocks. The body of the - kernel is defined by the single region that this operation contains. The - operation takes an optional list of async dependencies followed by six - operands and an optional operand. + kernel is defined by the single region that this operation contains. - The `async` keyword indicates the kernel should be launched asynchronously; - the operation returns a new !gpu.async.token when the keyword is specified. - The kernel launched does not start executing until the ops producing its - async dependencies (optional operands) have completed. + The async execution model is equivalent to the `gpu.launch_func` op, refer + to its description. The first three operands (following any async dependencies) are grid sizes along the x,y,z dimensions and the following three are block sizes along the @@ -945,6 +951,7 @@ def GPU_LaunchOp : GPU_Op<"launch", [ CArg<"Value", "nullptr">:$dynamicSharedMemorySize, CArg<"Type", "nullptr">:$asyncTokenType, CArg<"ValueRange", "{}">:$asyncDependencies, + CArg<"Value", "nullptr">:$asyncObject, CArg<"TypeRange", "{}">:$workgroupAttributions, CArg<"TypeRange", "{}">:$privateAttributions, CArg<"Value", "nullptr">:$clusterSizeX, @@ -1209,7 +1216,7 @@ def GPU_AllReduceOp : GPU_Op<"all_reduce", let results = (outs AnyIntegerOrFloat:$result); let regions = (region AnyRegion:$body); - let assemblyFormat = [{ custom($op) $value + let assemblyFormat = [{ (enum($op)^)? $value (`uniform` $uniform^)? $body attr-dict `:` functional-type(operands, results) }]; @@ -1288,10 +1295,9 @@ def GPU_SubgroupReduceOp : GPU_Op<"subgroup_reduce", [SameOperandsAndResultType, }]> ]; - let assemblyFormat = [{ custom($op) $value + let assemblyFormat = [{ enum($op) $value (`uniform` $uniform^)? - (`cluster` `(` `size` `=` $cluster_size^ (`,` `stride` `=` $cluster_stride^)? `)`)? - attr-dict + (`cluster` `(` `size` `=` $cluster_size^ (`,` `stride` `=` $cluster_stride^)? `)`)? attr-dict `:` functional-type(operands, results) }]; let hasFolder = 1; @@ -1373,7 +1379,7 @@ def GPU_ShuffleOp : GPU_Op< }]; let assemblyFormat = [{ - $mode $value `,` $offset `,` $width attr-dict `:` type($value) + enum($mode) $value `,` $offset `,` $width attr-dict `:` type($value) }]; let builders = [ @@ -1427,13 +1433,21 @@ def GPU_RotateOp : GPU_Op< } def GPU_BarrierOp : GPU_Op<"barrier">, - Arguments<(ins OptionalAttr :$address_spaces)> { - let summary = "Synchronizes all work items of a workgroup."; + Arguments<(ins + OptionalAttr:$address_spaces, + Optional:$named_barrier, + DefaultValuedAttr:$scope + )> { + let summary = "Synchronizes work items within an execution scope."; let description = [{ - The `barrier` op synchronizes all work items of a workgroup. It is used - to coordinate communication between the work items of the workgroup. + The `barrier` op synchronizes work items within the specified execution + scope. By default, the scope is `workgroup`, synchronizing all work items + in a workgroup. ```mlir + // Synchronize all work items in the workgroup, making all prior + // memory accesses visible. gpu.barrier ``` @@ -1443,17 +1457,35 @@ def GPU_BarrierOp : GPU_Op<"barrier">, accessing the same memory can be avoided by synchronizing work items in-between these accesses. - If the `memfence` attribute is specified, the set of memory accesses that must - by completed after the barrier resolves is limited to only those accesses that - read from or write to the specified address spaces (though accesses to other - address spaces may be completed as well, especially if a particular combination - of address spaces is not supported on a given backend). In particular, - specifying `memfence []` creates a barrier that is not required to affect - the visibility of any memory operations and is purely used for synchronizing - work items. + The `scope` attribute controls the execution scope of the barrier: + + ```mlir + // Synchronize within a subgroup (warp/wavefront). + gpu.barrier scope + // Synchronize within a cluster. + gpu.barrier scope + ``` + + A `named` barrier allows synchronizing a specific subset of subgroups + that have been associated with a named barrier handle. Named barriers + require workgroup scope. + + ```mlir + // Initialize a named barrier for 4 participating members. + %nb = gpu.initialize_named_barrier %c4 : i32 -> !gpu.named_barrier + // Wait on the named barrier. + gpu.barrier named(%nb : !gpu.named_barrier) + ``` + + If the `memfence` attribute is specified, the set of memory accesses that + must be completed after the barrier resolves is limited to only those + accesses that read from or write to the specified address spaces. In + particular, specifying `memfence []` creates a barrier that is not required + to affect the visibility of any memory operations and is purely used for + synchronizing work items. ```mlir - // Only workgroup address spaces accesses required to be visible. + // Only workgroup address space accesses required to be visible. gpu.barrier memfence [#gpu.address_space] // No memory accesses required to be visible. gpu.barrier memfence [] @@ -1461,20 +1493,62 @@ def GPU_BarrierOp : GPU_Op<"barrier">, gpu.barrier ``` - Either none or all work items of a workgroup need to execute this op - in convergence. + The three clauses can be combined in any order, but not all combinations may + be supported on a given target: + + ```mlir + // Named barrier with a workgroup-only memory fence. + gpu.barrier named(%nb : !gpu.named_barrier) memfence [#gpu.address_space] + // Subgroup barrier with a global fence. + gpu.barrier memfence [#gpu.address_space] scope + ``` + + Once one thread of execution in a given scope (say, thread in a workgroup) + has executed a particular dynamic instance of `gpu.barrier`, all other threads + in that scope are required to execute the same dynamic instance of `gpu.barrier` + before any thread executes any other instance of it. That is, you cannot, for + example, have the two subgroups of a workgroup arrive at `gpu.barrier` ops in + different branches of an if statement and have this work. + }]; + let assemblyFormat = [{ + oilist( + `named` `(` $named_barrier `:` type($named_barrier) `)` + | `memfence` $address_spaces + | `scope` $scope + ) attr-dict }]; - let assemblyFormat = "(`memfence` $address_spaces^)? attr-dict"; let hasCanonicalizer = 1; + let hasVerifier = 1; let builders = [OpBuilder<( ins CArg<"std::optional<::mlir::gpu::AddressSpace>", "std::nullopt">:$addressSpace)>, OpBuilder<(ins "Value":$memrefToFence)>]; } +def GPU_InitializeNamedBarrierOp + : GPU_Op<"initialize_named_barrier", + [MemoryEffects<[MemAlloc]>]> { + let summary = "Initialize a named barrier with a member count."; + let description = [{ + Initializes a named barrier object with the given number of participating + members (subgroups) and returns a handle to it. All members that will + synchronize on this barrier must be accounted for in the count. + + ```mlir + %nb = gpu.initialize_named_barrier %num_members : i32 -> !gpu.named_barrier + ``` + }]; + let arguments = (ins I32:$member_count); + let results = (outs GPU_NamedBarrier:$result); + let assemblyFormat = [{ + $member_count attr-dict `:` type($member_count) `->` type($result) + }]; +} + def GPU_GPUModuleOp : GPU_Op<"module", [ IsolatedFromAbove, DataLayoutOpInterface, HasDefaultDLTIDataLayout, - NoRegionArguments, SymbolTable, Symbol] # GraphRegionNoTerminator.traits> { + NoRegionArguments, SymbolTable, SymbolName, SymbolVisibility, + Symbol] # GraphRegionNoTerminator.traits> { let summary = "A top level compilation unit containing code to be run on a GPU."; let description = [{ GPU module contains code that is intended to be run on a GPU. A host device @@ -1524,14 +1598,14 @@ def GPU_GPUModuleOp : GPU_Op<"module", [ let arguments = (ins SymbolNameAttr:$sym_name, + OptionalAttr:$sym_visibility, OptionalAttr:$targets, OptionalAttr:$offloadingHandler); let regions = (region SizedRegion<1>:$bodyRegion); let assemblyFormat = [{ - $sym_name + ($sym_visibility^)? $sym_name (`<` $offloadingHandler^ `>`)? - ($targets^)? - attr-dict-with-keyword $bodyRegion + ($targets^)? attr-dict-with-keyword $bodyRegion }]; // We need to ensure the block inside the region is properly terminated; @@ -1549,8 +1623,10 @@ def GPU_GPUModuleOp : GPU_Op<"module", [ let hasVerifier = 1; } -def GPU_BinaryOp : GPU_Op<"binary", [Symbol]>, Arguments<(ins +def GPU_BinaryOp + : GPU_Op<"binary", [SymbolName, SymbolVisibility, Symbol]>, Arguments<(ins SymbolNameAttr:$sym_name, + OptionalAttr:$sym_visibility, OptionalAttr:$offloadingHandler, ConfinedAttr]>:$objects) > { @@ -1589,7 +1665,8 @@ def GPU_BinaryOp : GPU_Op<"binary", [Symbol]>, Arguments<(ins ]; let skipDefaultBuilders = 1; let assemblyFormat = [{ - $sym_name custom($offloadingHandler) attr-dict $objects + ($sym_visibility^)? $sym_name + custom($offloadingHandler) attr-dict $objects }]; } @@ -1872,7 +1949,9 @@ def GPU_SubgroupMmaLoadMatrixOp : GPU_Op<"subgroup_mma_load_matrix", let results = (outs GPU_MMAMatrix:$res); let assemblyFormat = [{ - $srcMemref`[`$indices`]` attr-dict `:` type($srcMemref) `->` type($res) + $srcMemref`[`$indices`]` `leadDimension` $leadDimension + (`transpose` $transpose^)? attr-dict `:` type($srcMemref) `->` + type($res) }]; let hasVerifier = 1; } @@ -1915,7 +1994,9 @@ def GPU_SubgroupMmaStoreMatrixOp : GPU_Op<"subgroup_mma_store_matrix", OptionalAttr:$transpose); let assemblyFormat = [{ - $src`,` $dstMemref`[`$indices`]` attr-dict `:` type($src)`,` type($dstMemref) + $src`,` $dstMemref`[`$indices`]` `leadDimension` $leadDimension + (`transpose` $transpose^)? attr-dict `:` type($src)`,` + type($dstMemref) }]; let hasVerifier = 1; } @@ -1964,7 +2045,9 @@ def GPU_SubgroupMmaComputeOp let results = (outs GPU_MMAMatrix : $res); let assemblyFormat = [{ - $opA`,` $opB`,` $opC attr-dict `:` type($opA)`,` type($opB) `->` type($res) + $opA`,` $opB`,` $opC (`a_transpose` $a_transpose^)? + (`b_transpose` $b_transpose^)? attr-dict `:` type($opA)`,` + type($opB) `->` type($res) }]; let hasVerifier = 1; } @@ -2180,7 +2263,7 @@ def GPU_SubgroupMmaElementwiseOp : GPU_Op<"subgroup_mma_elementwise", }]; let assemblyFormat = [{ - $opType $args attr-dict `:` functional-type($args, $res) + enum($opType) $args attr-dict `:` functional-type($args, $res) }]; } @@ -2500,7 +2583,7 @@ def GPU_Create2To4SpMatOp : GPU_Op<"create_2to4_spmat", [GPU_AsyncOpInterface]> let assemblyFormat = [{ custom(type($asyncToken), $asyncDependencies) - `{` $pruneFlag `}` $rows `,` $cols `,` $memref attr-dict `:` type($memref) + `{` enum($pruneFlag) `}` $rows `,` $cols `,` $memref attr-dict `:` type($memref) }]; } @@ -2603,7 +2686,7 @@ def GPU_SpMVBufferSizeOp : GPU_Op<"spmv_buffer_size", [GPU_AsyncOpInterface]> { let assemblyFormat = [{ custom(type($asyncToken), $asyncDependencies) - $spmatA (`{` $modeA^ `}`)? `,` $dnX `,` $dnY attr-dict `into` $computeType + $spmatA (`{` enum($modeA)^ `}`)? `,` $dnX `,` $dnY attr-dict `into` $computeType }]; } @@ -2653,7 +2736,7 @@ def GPU_SpMVOp : GPU_Op<"spmv", [GPU_AsyncOpInterface]> { let assemblyFormat = [{ custom(type($asyncToken), $asyncDependencies) - $spmatA (`{` $modeA^ `}`)? `,` $dnX `,` $dnY `,` $buffer attr-dict `:` type($buffer) `into` $computeType + $spmatA (`{` enum($modeA)^ `}`)? `,` $dnX `,` $dnY `,` $buffer attr-dict `:` type($buffer) `into` $computeType }]; } @@ -2706,7 +2789,9 @@ def GPU_SpMMBufferSizeOp : GPU_Op<"spmm_buffer_size", [GPU_AsyncOpInterface, Att let assemblyFormat = [{ custom(type($asyncToken), $asyncDependencies) - $spmatA (`{` $modeA^ `}`)? `,` $dnmatB (`{` $modeB^ `}`)? `,` $dnmatC attr-dict `:` type($bufferSzs) `into` $computeType + $spmatA (`{` enum($modeA)^ `}`)? `,` $dnmatB + (`{` enum($modeB)^ `}`)? `,` $dnmatC attr-dict `:` type($bufferSzs) + `into` $computeType }]; } @@ -2759,7 +2844,9 @@ def GPU_SpMMOp : GPU_Op<"spmm", [GPU_AsyncOpInterface, AttrSizedOperandSegments] let assemblyFormat = [{ custom(type($asyncToken), $asyncDependencies) - $spmatA (`{` $modeA^ `}`)? `,` $dnmatB (`{` $modeB^ `}`)? `,` $dnmatC `,` $buffers attr-dict `:` type($buffers) `into` $computeType + $spmatA (`{` enum($modeA)^ `}`)? `,` $dnmatB + (`{` enum($modeB)^ `}`)? `,` $dnmatC `,` $buffers attr-dict `:` + type($buffers) `into` $computeType }]; } @@ -2811,7 +2898,8 @@ def GPU_SDDMMBufferSizeOp : GPU_Op<"sddmm_buffer_size", [GPU_AsyncOpInterface]> let assemblyFormat = [{ custom(type($asyncToken), $asyncDependencies) - $dnmatA (`{` $modeA^ `}`)? `,` $dnmatB (`{` $modeB^ `}`)? `,` $spmatC attr-dict `into` $computeType + $dnmatA (`{` enum($modeA)^ `}`)? `,` $dnmatB + (`{` enum($modeB)^ `}`)? `,` $spmatC attr-dict `into` $computeType }]; } @@ -2864,7 +2952,9 @@ def GPU_SDDMMOp : GPU_Op<"sddmm", [GPU_AsyncOpInterface]> { let assemblyFormat = [{ custom(type($asyncToken), $asyncDependencies) - $dnmatA (`{` $modeA^ `}`)? `,` $dnmatB (`{` $modeB^ `}`)? `,` $spmatC `,` $buffer attr-dict `:` type($buffer) `into` $computeType + $dnmatA (`{` enum($modeA)^ `}`)? `,` $dnmatB + (`{` enum($modeB)^ `}`)? `,` $spmatC `,` $buffer attr-dict `:` + type($buffer) `into` $computeType }]; } @@ -2904,8 +2994,7 @@ def GPU_SpGEMMCreateDescrOp : GPU_Op<"spgemm_create_descr", [GPU_AsyncOpInterfac let results = (outs GPU_SparseSpGEMMOpHandle:$desc, Optional:$asyncToken); let assemblyFormat = [{ - custom(type($asyncToken), $asyncDependencies) - attr-dict + custom(type($asyncToken), $asyncDependencies) attr-dict }]; } @@ -2998,7 +3087,9 @@ def GPU_SpGEMMWorkEstimationOrComputeOp : GPU_Op<"spgemm_work_estimation_or_comp let assemblyFormat = [{ custom(type($asyncToken), $asyncDependencies) - `{` $kind `}` $spmatA (`{` $modeA^ `}`)? `,` $spmatB (`{` $modeB^ `}`)? `,` $spmatC `,` $desc `,` $bufferSz `,` $buffer attr-dict `:` $computeType `into` type($buffer) + `{` enum($kind) `}` $spmatA (`{` enum($modeA)^ `}`)? `,` $spmatB + (`{` enum($modeB)^ `}`)? `,` $spmatC `,` $desc `,` $bufferSz `,` + $buffer attr-dict `:` $computeType `into` type($buffer) }]; } @@ -3049,7 +3140,8 @@ def GPU_SpGEMMCopyOp : GPU_Op<"spgemm_copy", [GPU_AsyncOpInterface]> { let assemblyFormat = [{ custom(type($asyncToken), $asyncDependencies) - $spmatA (`{` $modeA^ `}`)? `,` $spmatB (`{` $modeB^ `}`)? `,` $spmatC `,` $desc attr-dict `:` $computeType + $spmatA (`{` enum($modeA)^ `}`)? `,` $spmatB + (`{` enum($modeB)^ `}`)? `,` $spmatC `,` $desc attr-dict `:` $computeType }]; } @@ -3289,7 +3381,7 @@ def GPU_SubgroupBroadcastOp : GPU_Op<"subgroup_broadcast", }]; let results = (outs AnyType:$result); let assemblyFormat = [{ - $src `,` $broadcast_type ($lane^)? attr-dict `:` type($result) + $src `,` enum($broadcast_type) ($lane^)? attr-dict `:` type($result) }]; let hasFolder = 1; let hasVerifier = 1; diff --git a/examples/mlir/Dialect/Linalg/LinalgEnums.td b/examples/mlir/Dialect/Linalg/LinalgEnums.td index 1109db9..d18731c 100644 --- a/examples/mlir/Dialect/Linalg/LinalgEnums.td +++ b/examples/mlir/Dialect/Linalg/LinalgEnums.td @@ -29,7 +29,19 @@ def UnaryFn : I32EnumAttr<"UnaryFn", "", [ I32EnumAttrCase<"rsqrt", 9>, I32EnumAttrCase<"square", 10>, I32EnumAttrCase<"tanh", 11>, - I32EnumAttrCase<"erf", 12> + I32EnumAttrCase<"erf", 12>, + I32EnumAttrCase<"sin", 13>, + I32EnumAttrCase<"cos", 14>, + I32EnumAttrCase<"tan", 15>, + I32EnumAttrCase<"acos", 16>, + I32EnumAttrCase<"acosh", 17>, + I32EnumAttrCase<"asin", 18>, + I32EnumAttrCase<"asinh", 19>, + I32EnumAttrCase<"atan", 20>, + I32EnumAttrCase<"atanh", 21>, + I32EnumAttrCase<"log10", 22>, + I32EnumAttrCase<"log1p", 23>, + I32EnumAttrCase<"log2", 24> ]> { let genSpecializedAttr = 0; let cppNamespace = "::mlir::linalg"; diff --git a/examples/mlir/Dialect/Linalg/LinalgStructuredOps.td b/examples/mlir/Dialect/Linalg/LinalgStructuredOps.td index 5998f73..2b6fc04 100644 --- a/examples/mlir/Dialect/Linalg/LinalgStructuredOps.td +++ b/examples/mlir/Dialect/Linalg/LinalgStructuredOps.td @@ -62,14 +62,13 @@ def GenericOp : LinalgStructuredBase_Op<"generic", [ ```mlir linalg.generic #trait_attribute - ins(%A, %B : memref, - memref) - outs(%C : memref) + ins(%A, %B : memref, memref) + outs(%C : memref) attrs = {other-optional-attributes} {region} ``` - Where #trait_attributes is an alias of a dictionary attribute containing: + Where #trait_attribute is an alias of a dictionary attribute containing: - doc [optional]: a documentation string - indexing_maps: a list of AffineMapAttr, one AffineMapAttr per each input and output view. Such AffineMapAttr specifies the mapping between the @@ -80,17 +79,17 @@ def GenericOp : LinalgStructuredBase_Op<"generic", [ compile-time guarantees are provided. In the absence of such a library call, linalg.generic will always lower to loops. - iterator_types: an ArrayAttr specifying the type of the enclosing loops. - Each element of the list represents and iterator of one of the following + Each element of the list represents an iterator of one of the following types: - parallel, reduction, window + parallel, reduction Example: Defining a #matmul_trait attribute in MLIR can be done as follows: ```mlir #matmul_accesses = [ - (m, n, k) -> (m, k), - (m, n, k) -> (k, n), - (m, n, k) -> (m, n) + affine_map<(m, n, k) -> (m, k)>, + affine_map<(m, n, k) -> (k, n)>, + affine_map<(m, n, k) -> (m, n)> ] #matmul_trait = { doc = "C(m, n) += A(m, k) * B(k, n)", @@ -103,10 +102,9 @@ def GenericOp : LinalgStructuredBase_Op<"generic", [ And can be reused in multiple places as: ```mlir linalg.generic #matmul_trait - ins(%A, %B : memref, - memref) - outs(%C : memref) - {other-optional-attributes} { + ins(%A, %B : memref, memref) + outs(%C : memref) + attrs = {other-optional-attributes} { ^bb0(%a: f32, %b: f32, %c: f32) : %d = arith.mulf %a, %b: f32 %e = arith.addf %c, %d: f32 @@ -117,9 +115,9 @@ def GenericOp : LinalgStructuredBase_Op<"generic", [ This may lower to either: ```mlir call @linalg_matmul(%A, %B, %C) : - (memref, - memref, - memref) + (memref>, + memref>, + memref>) -> () ``` @@ -128,12 +126,12 @@ def GenericOp : LinalgStructuredBase_Op<"generic", [ scf.for %m = %c0 to %M step %c1 { scf.for %n = %c0 to %N step %c1 { scf.for %k = %c0 to %K step %c1 { - %a = load %A[%m, %k] : memref - %b = load %B[%k, %n] : memref - %c = load %C[%m, %n] : memref + %a = memref.load %A[%m, %k] : memref + %b = memref.load %B[%k, %n] : memref + %c = memref.load %C[%m, %n] : memref %d = arith.mulf %a, %b: f32 %e = arith.addf %c, %d: f32 - store %e, %C[%m, %n] : memref + memref.store %e, %C[%m, %n] : memref } } } @@ -388,6 +386,7 @@ def ReduceOp : LinalgStructuredBase_Op<"reduce", [ }]; let hasCustomAssemblyFormat = 1; + let hasCanonicalizer = 1; let hasVerifier = 1; } @@ -478,6 +477,11 @@ def BroadcastOp : LinalgStructuredBase_Op<"broadcast", [ let description = [{ Broadcast the input into the given shape by adding `dimensions`. + Each index in the `dimensions` attribute refers to a dimension of `init` + that is added by the operation. The indices must be unique and within the + rank of `init`; the sizes of the remaining (non-added) dimensions of + `init` must match the shape of `input`. + Example: ```mlir %bcast = linalg.broadcast @@ -569,16 +573,14 @@ def ElementwiseOp : LinalgStructuredBase_Op<"elementwise", [ Defining a unary linalg.elementwise with default indexing-map: ```mlir - %exp = linalg.elementwise - kind=#linalg.elementwise_kind + %exp = linalg.elementwise ins(%x : tensor<4x16x8xf32>) outs(%y: tensor<4x16x8xf32>) -> tensor<4x16x8xf32> ``` Defining a binary linalg.elementwise with user-defined indexing-map: ```mlir - %add = linalg.elementwise - kind=#linalg.elementwise_kind + %add = linalg.elementwise indexing_maps = [#transpose, #broadcast, #identity] ins(%exp, %arg1 : tensor<4x16x8xf32>, tensor<4x16xf32>) outs(%arg2: tensor<4x8x16xf32>) -> tensor<4x8x16xf32> @@ -606,11 +608,21 @@ def ElementwiseOp : LinalgStructuredBase_Op<"elementwise", [ }]>, OpBuilder<(ins "ValueRange":$inputs, "ValueRange":$outputs, - "ElementwiseKindAttr":$kind, - "ArrayAttr":$indexingMaps, + "ElementwiseKind":$kind, + CArg<"ArrayAttr", "{}">:$indexingMaps, CArg<"ArrayRef", "{}">:$attributes), [{ - $_state.addAttribute("kind", kind); + assert((unsigned)kind <= getMaxEnumValForElementwiseKind() && + "expected a valid elementwise kind attribute"); + ElementwiseKindAttr kindAttr = ElementwiseKindAttr::get($_builder.getContext(), kind); + $_state.addAttribute("kind", kindAttr); + if (!indexingMaps) { + auto affineMaps = ElementwiseOp::getDefaultIndexingMaps( + inputs.size() + outputs.size(), + llvm::cast(outputs[0].getType()).getRank(), + $_builder.getContext()); + indexingMaps = $_builder.getAffineMapArrayAttr(affineMaps); + } $_state.addAttribute("indexing_maps", indexingMaps); buildStructuredOp($_builder, $_state, std::nullopt, inputs, outputs, attributes, ElementwiseOp::getRegionBuilder()); @@ -881,7 +893,7 @@ def ContractOp : LinalgStructuredBase_Op<"contract", [ AffineMapArrayAttr:$indexing_maps, DefaultValuedOptionalAttr:$cast ); - let results = (outs Variadic:$result_tensors); + let results = (outs Variadic:$result_tensors); // NB: The only reason this op has a region - and it get populated at op build // time - is that currently the LinalgOp interface exposes methods that // assume a relevant region is available to be queried at any time. @@ -1213,6 +1225,169 @@ def BatchReduceMatmulOp : LinalgStructuredBase_Op<"batch_reduce_matmul", [ }]; } +//===----------------------------------------------------------------------===// +// Scaled Contract op. +//===----------------------------------------------------------------------===// + +def ScaledContractOp : LinalgStructuredBase_Op<"scaled_contract", [ + AttrSizedOperandSegments]> { + let summary = [{ + Perform a scaled contraction on two inputs where their scaling is described + by affine maps of the two corresponding scales, accumulating into the output. + }]; + let description = [{ + Extends the semantics of `linalg.contract` by including extra scaling factors + for the inputs `A` and `B` described by the scales' affine maps. + The data types of the inputs are chosen independently from the scales. + + The output element type must be floating-point. Input element types must be + integer or floating-point, with bitwidth no greater than the output element + type's bitwidth. + + The semantics of contracting inputs `A` and `B` with scales `scale_A` and + `scale_B` on top of `C` to produce output `D` is given by: + + `D[H] = (SUM_{(I ∪ J) \ H} (A[I] * scale_A[I]) * (B[J] * scale_B[J])) + C[H]` + + The iteration type of each dim is inferred. + + The input operands `A`, `B`, and `C` follow the standard contraction semantics + together with broadcasting and transposition rules. + See `linalg.contract` for further details. + + **Affine maps for scales** + + Each scale indexing map describes the scaling scheme for the corresponding + input `A` and `B`. The following scaling schemes are supported: + + - Tensor scaling - no dimensions: a scalar value used to scale the whole input. + - Dimension scaling - dimension `d`: one scale value per element along `d`. + - Block scaling - binary expression `d floordiv B`: one scale value for every + `B` elements along `d` where `B` corresponds to the block size. + + Scale maps may only use dimensions present in the corresponding input's map. + A dimension absent from the scale map indicates that a single scale value is + reused across the whole input dimension. + + **Dim sizes for scales** + + - Tensor scaling: a unit-size scale. + - Dimension scaling (`d`): the scale dim size equals the input dim size. + - Block scaling (`d floordiv B`): the scale dim size equals + `ceil(input_dim_size / B)`. + + For dynamic shapes, all sizes are assumed to be correct at runtime. + + Example scaled matmul with transposed B using semi-affine maps: + ```mlir + %D = linalg.scaled_contract + indexing_maps = [ + affine_map<(m, n, k) -> (m, k)>, + affine_map<(m, n, k) -> (m floordiv 32, k floordiv 128)>, // A - 32x128 block scale + affine_map<(m, n, k) -> (n, k)>, + affine_map<(m, n, k) -> (n)>, // B - row-wise scale + affine_map<(m, n, k) -> (m, n)>] + ins(%A, %scale_A, %B, %scale_B + : tensor<200x512xf8E5M2>, tensor<7x4xf8E8M0FNU>, + tensor<128x512xf8E5M2>, tensor<128xf8E8M0FNU>) + outs(%C: tensor<200x128xf32>) -> tensor<200x128xf32> + ``` + + or using expanded dimensions: + + ```mlir + %D = linalg.scaled_contract + indexing_maps = [ + affine_map<(m, n, ko, kb) -> (m, ko, kb)>, // A - 32 block dim 'kb' + affine_map<(m, n, ko, kb) -> (m, ko)>, + affine_map<(m, n, ko, kb) -> (n, ko, kb)>, // B - 32 block dim 'kb' + affine_map<(m, n, ko, kb) -> (n, ko)>, + affine_map<(m, n, ko, kb) -> (m, n)>] + ins(%A, %scale_A, %B, %scale_B + : tensor<128x4x32xf4E2M1FN>, tensor<128x4xf8E8M0FNU>, + tensor<64x4x32xf4E2M1FN>, tensor<64x4xf8E8M0FNU>) + outs(%C : tensor<128x64xf32>) -> tensor<128x64xf32> + ``` + + Numeric casting promotes inputs and scales to the accumulator/output type. + + TODO: Allow control over the combining/accumulating op and possibly the + multiplication op. + }]; + + let arguments = (ins + Variadic:$inputs, + Variadic:$outputs, + AffineMapArrayAttr:$indexing_maps, + DefaultValuedOptionalAttr:$cast + ); + let results = (outs Variadic:$result_tensors); + let regions = (region SizedRegion<1>:$combiner); + + let skipDefaultBuilders = 1; + let builders = [ + OpBuilder< + (ins "ValueRange":$inputs, "ValueRange":$outputs, + CArg<"ArrayRef", "{}">:$attributes), + [{ + buildStructuredOp($_builder, $_state, std::nullopt, inputs, outputs, + attributes, regionBuilder); + }]>, + OpBuilder<(ins "TypeRange":$resultTensorTypes, "ValueRange":$inputs, + "ValueRange":$outputs, "ArrayAttr":$indexingMaps, + CArg<"ArrayRef", "{}">:$attributes), + [{ + $_state.addAttribute("indexing_maps", indexingMaps); + buildStructuredOp($_builder, $_state, resultTensorTypes, inputs, + outputs, attributes, regionBuilder); + }]>, + OpBuilder<(ins "ValueRange":$inputs, "ValueRange":$outputs, + "ArrayAttr":$indexingMaps, + CArg<"ArrayRef", "{}">:$attributes), + [{ + $_state.addAttribute("indexing_maps", indexingMaps); + buildStructuredOp($_builder, $_state, std::nullopt, inputs, outputs, + attributes, regionBuilder); + }]> + ]; + let hasCustomAssemblyFormat = 1; + let hasFolder = 1; + let hasVerifier = 1; + + let extraClassDeclaration = structuredOpsBaseDecls # [{ + // Declare/implement functions necessary for LinalgStructuredInterface. + + /// Infer iterator types for each dim in the domain of IndexingMaps. + SmallVector getIteratorTypesArray(); + + /// IndexingMaps always depends on attr associated to current Op instance. + bool hasDynamicIndexingMaps() { return true; }; + bool hasUserDefinedMaps() { return true; }; + + static unsigned getNumRegionArgs(); + + static void regionBuilder(ImplicitLocOpBuilder &b, + Block &block, ArrayRef attrs, + function_ref emitError); + + static std::function, + function_ref)> + getRegionBuilder() { + return regionBuilder; + } + + std::string getLibraryCallName() { + return "op_has_no_registered_library_name"; + } + + // Implement function necessary for DestinationStyleOpInterface. + ::mlir::MutableOperandRange getDpsInitsMutable() { + return getOutputsMutable(); + } + }]; +} + //===----------------------------------------------------------------------===// // Named Linalg ops, implemented as a declarative configurations of generic ops. //===----------------------------------------------------------------------===// diff --git a/examples/mlir/Dialect/MemRef/MemRefOps.td b/examples/mlir/Dialect/MemRef/MemRefOps.td index 9dba4d7..86e466c 100644 --- a/examples/mlir/Dialect/MemRef/MemRefOps.td +++ b/examples/mlir/Dialect/MemRef/MemRefOps.td @@ -134,7 +134,8 @@ class AllocLikeOp ((d0 + s0), d1)>, 1> ``` }]; @@ -320,7 +321,7 @@ def MemRef_ReallocOp : MemRef_Op<"realloc", aligned_realloc. ```mlir - %3 = memref.realloc %src {alignment = 8} : memref<64xf32> to memref<124xf32> + %3 = memref.realloc %src alignment = 8 : memref<64xf32> to memref<124xf32> ``` Referencing the memref through the old SSA value after realloc is undefined @@ -360,7 +361,8 @@ def MemRef_ReallocOp : MemRef_Op<"realloc", }]; let assemblyFormat = [{ - $source (`(` $dynamicResultSize^ `)`)? attr-dict + $source (`(` $dynamicResultSize^ `)`)? + (`alignment` `=` $alignment^)? attr-dict `:` type($source) `to` type(results) }]; @@ -1188,7 +1190,7 @@ def MemRef_GetGlobalOp : MemRef_Op<"get_global", // GlobalOp //===----------------------------------------------------------------------===// -def MemRef_GlobalOp : MemRef_Op<"global", [Symbol, +def MemRef_GlobalOp : MemRef_Op<"global", [SymbolName, SymbolVisibility, Symbol, DeclareOpInterfaceMethods]> { let summary = "declare or define a global memref variable"; let description = [{ @@ -1215,7 +1217,7 @@ def MemRef_GlobalOp : MemRef_Op<"global", [Symbol, memref.global "private" @x : memref<2xf32> = dense<[0.0, 2.0]> // Private variable with an initial value and an alignment (power of 2). - memref.global "private" @x : memref<2xf32> = dense<[0.0, 2.0]> {alignment = 64} + memref.global "private" @x : memref<2xf32> = dense<[0.0, 2.0]> alignment = 64 // Declaration of an external variable. memref.global "private" @y : memref<4xi32> @@ -1240,7 +1242,7 @@ def MemRef_GlobalOp : MemRef_Op<"global", [Symbol, (`constant` $constant^)? $sym_name `:` custom($type, $initial_value) - attr-dict + (`alignment` `=` $alignment^)? attr-dict }]; let extraClassDeclaration = [{ @@ -1276,10 +1278,14 @@ def LoadOp : MemRef_Op<"load", The number of indices must match the rank of the memref. The indices must be in-bounds: `0 <= idx < dim_size`. - Lowerings of `memref.load` may emit attributes, e.g. `inbouds` + `nuw` - when converting to LLVM's `llvm.getelementptr`, that would cause undefined - behavior if indices are out of bounds or if computing the offset in the - memref would cause signed overflow of the `index` type. + Lowerings of `memref.load` may emit no-wrap flags on + `llvm.getelementptr` when converting to LLVM. The `inbounds` flag is + always emitted (valid since indices are guaranteed in-bounds) and causes + undefined behavior if that precondition is violated. The `nuw` flag is + emitted only when all strides of the memref are statically non-negative; + with negative strides, `nuw` would propagate to intermediate `mul` + operations and cause unsigned overflow (poison) even for in-bounds + indices. The single result of `memref.load` is a value with the same type as the element type of the memref. @@ -1288,6 +1294,12 @@ def LoadOp : MemRef_Op<"load", be reused in the cache. For details, refer to the [LLVM load instruction](https://llvm.org/docs/LangRef.html#load-instruction). + A set `invariant` attribute indicates that the referenced memory location + contains the same value at all points in the program where it is + dereferenceable, so the load may be treated as invariant. For details, refer + to the + [LLVM load instruction](https://llvm.org/docs/LangRef.html#load-instruction). + An optional `alignment` attribute allows to specify the byte alignment of the load operation. It must be a positive power of 2. The operation must access memory at an address aligned to this boundary. Violations may lead to @@ -1304,7 +1316,8 @@ def LoadOp : MemRef_Op<"load", [MemRead]>:$memref, Variadic:$indices, DefaultValuedOptionalAttr:$nontemporal, - OptionalAttr>:$alignment); + OptionalAttr>:$alignment, + DefaultValuedOptionalAttr:$invariant); let builders = [ OpBuilder<(ins "Value":$memref, @@ -1347,7 +1360,15 @@ def LoadOp : MemRef_Op<"load", let hasFolder = 1; - let assemblyFormat = "$memref `[` $indices `]` attr-dict `:` type($memref)"; + let assemblyFormat = [{ + $memref `[` $indices `]` + oilist( + `alignment` `(` $alignment `)` | + `nontemporal` `(` custom($nontemporal) `)` | + `invariant` `(` custom($invariant) `)` + ) + attr-dict `:` type($memref) + }]; } //===----------------------------------------------------------------------===// @@ -1500,6 +1521,12 @@ def MemRef_ReinterpretCastOp Consecutive `reinterpret_cast` operations on memref's with static dimensions. + This operation is intended for cases where the user can guarantee the + validity of the constructed descriptor. Neither static nor runtime + verification check that the resulting descriptor is in-bounds. Accessing + memory outside the underlying allocation through the resulting memref is + undefined behavior. + We distinguish between *underlying memory* — the sequence of elements as they appear in the contiguous memory of the memref — and the *strided memref*, which refers to the underlying memory interpreted @@ -1756,15 +1783,7 @@ def MemRef_ReshapeOp: MemRef_Op<"reshape", [ "dynamically-sized shape", [MemRead]>:$shape); let results = (outs AnyRankedOrUnrankedMemRef:$result); - let builders = [OpBuilder< - (ins "MemRefType":$resultType, "Value":$operand, "Value":$shape), [{ - $_state.addOperands(operand); - $_state.addOperands(shape); - $_state.addTypes(resultType); - }]>]; - let extraClassDeclaration = [{ - MemRefType getType() { return ::llvm::cast(getResult().getType()); } Value getViewSource() { return getSource(); } }]; @@ -2008,9 +2027,11 @@ def MemRef_CollapseShapeOp : MemRef_ReassociativeReshapeOp<"collapse_shape", [ "ArrayRef":$reassociation, CArg<"ArrayRef", "{}">:$attrs), [{ - $_state.addAttribute("reassociation", - getReassociationIndicesAttribute($_builder, reassociation)); - build($_builder, $_state, resultType, src, attrs); + buildPropertiesAndDiscardableAttributes($_state, attrs); + $_state.getOrAddProperties().reassociation = + getReassociationIndicesAttribute($_builder, reassociation); + $_state.addOperands(src); + $_state.addTypes(resultType); }]>, OpBuilder<(ins "Type":$resultType, "Value":$src, "ArrayRef":$reassociation, @@ -2057,10 +2078,14 @@ def MemRef_StoreOp : MemRef_Op<"store", The number of indices must match the rank of the memref. The indices must be in-bounds: `0 <= idx < dim_size`. - Lowerings of `memref.store` may emit attributes, e.g. `inbouds` + `nuw` - when converting to LLVM's `llvm.getelementptr`, that would cause undefined - behavior if indices are out of bounds or if computing the offset in the - memref would cause signed overflow of the `index` type. + Lowerings of `memref.store` may emit no-wrap flags on + `llvm.getelementptr` when converting to LLVM. The `inbounds` flag is + always emitted (valid since indices are guaranteed in-bounds) and causes + undefined behavior if that precondition is violated. The `nuw` flag is + emitted only when all strides of the memref are statically non-negative; + with negative strides, `nuw` would propagate to intermediate `mul` + operations and cause unsigned overflow (poison) even for in-bounds + indices. A set `nontemporal` attribute indicates that this store is not expected to be reused in the cache. For details, refer to the @@ -2114,7 +2139,12 @@ def MemRef_StoreOp : MemRef_Op<"store", let hasFolder = 1; let assemblyFormat = [{ - $value `,` $memref `[` $indices `]` attr-dict `:` type($memref) + $value `,` $memref `[` $indices `]` + oilist( + `alignment` `(` $alignment `)` | + `nontemporal` `(` custom($nontemporal) `)` + ) + attr-dict `:` type($memref) }]; } @@ -2167,8 +2197,10 @@ def SubViewOp : MemRef_OpWithOffsetSizesAndStrides<"subview", [ The offset, size and stride operands must be in-bounds with respect to the source memref. When possible, the static operation verifier will detect out-of-bounds subviews. Subviews that cannot be confirmed to be in-bounds - or out-of-bounds based on compile-time information are valid. However, - performing an out-of-bounds subview at runtime is undefined behavior. + or out-of-bounds based on compile-time information are valid. The + `-generate-runtime-verification` pass can insert runtime bound checks. + Otherwise, performing an out-of-bounds subview at runtime is undefined + behavior. Example 1: @@ -2402,6 +2434,7 @@ def SubViewOp : MemRef_OpWithOffsetSizesAndStrides<"subview", [ def MemRef_TransposeOp : MemRef_Op<"transpose", [ DeclareOpInterfaceMethods, DeclareOpInterfaceMethods, + DeclareOpInterfaceMethods, Pure]>, Arguments<(ins AnyStridedMemRef:$in, AffineMapAttr:$permutation)>, Results<(outs AnyStridedMemRef)> { diff --git a/examples/mlir/Dialect/MemRef/MemoryAccessOpInterfaces.td b/examples/mlir/Dialect/MemRef/MemoryAccessOpInterfaces.td index 7fc69b4..b599eca 100644 --- a/examples/mlir/Dialect/MemRef/MemoryAccessOpInterfaces.td +++ b/examples/mlir/Dialect/MemRef/MemoryAccessOpInterfaces.td @@ -34,8 +34,8 @@ def IndexedAccessOpInterface : OpInterface<"IndexedAccessOpInterface"> { let methods = [InterfaceMethod< /*desc=*/[{ - Return the accessed memref. If the operation is still in tensor form, return - the null value. + Return the accessed memref. If the operation is not accessing memory through + a memref in its current form, return the null value. }], /*retType=*/"::mlir::TypedValue<::mlir::MemRefType>", /*methodName=*/"getAccessedMemref", @@ -53,7 +53,11 @@ def IndexedAccessOpInterface : OpInterface<"IndexedAccessOpInterface"> { InterfaceMethod< /*desc=*/[{ Return the shape of the portion of the memref that is being accessed by - this operation, if known, ignoring leading unit dimensions. + this operation, if known. This shape describes the access dimensions + whose strides are semantically important for this operation. + Implementations shall omit dimensions whose strides do not affect the + operation semantics. (In particular, if an operation will access one + element of the base memref, this method should return `{}`.) Reindexing transformations may not modify the *strides* of the trailing N dimensions, where N is the size returned value, and should ensure that @@ -136,7 +140,8 @@ def IndexedMemCopyOpInterface : OpInterface<"IndexedMemCopyOpInterface"> { let methods = [InterfaceMethod< /*desc=*/[{ - Return the source memref for this copy operation. + Return the source memref for this copy operation. If the operation is not + currently copying from a memref source, return the null value. }], /*retType=*/"::mlir::TypedValue<::mlir::MemRefType>", /*methodName=*/"getSrc", @@ -153,7 +158,8 @@ def IndexedMemCopyOpInterface : OpInterface<"IndexedMemCopyOpInterface"> { /*args=*/(ins)>, InterfaceMethod< /*desc=*/[{ - Return the destination memref for this copy operation. + Return the destination memref for this copy operation. If the operation is + not currently copying to a memref destination, return the null value. }], /*retType=*/"::mlir::TypedValue<::mlir::MemRefType>", /*methodName=*/"getDst", @@ -190,7 +196,33 @@ def IndexedMemCopyOpInterface : OpInterface<"IndexedMemCopyOpInterface"> { /*methodName=*/"setMemrefsAndIndices", /*args=*/(ins "::mlir::RewriterBase&":$rewriter, "::mlir::Value":$newSrc, "::mlir::ValueRange":$newSrcIndices, "::mlir::Value":$newDst, - "::mlir::ValueRange":$newDstIndices)>]; + "::mlir::ValueRange":$newDstIndices)>, + InterfaceMethod< + /*desc=*/[{ + Return true if, either by definition or due to some attribute, it's + known that the source indices are non-negative and less than the size of + the dimension they index. + }], + /*retType=*/"bool", + /*methodName=*/"hasInboundsSrcIndices", + /*args=*/(ins), + /*methodBody=*/[{}], + /*defaultImplementation=*/[{ + return true; + }]>, + InterfaceMethod< + /*desc=*/[{ + Return true if, either by definition or due to some attribute, it's + known that the destination indices are non-negative and less than the + size of the dimension they index. + }], + /*retType=*/"bool", + /*methodName=*/"hasInboundsDstIndices", + /*args=*/(ins), + /*methodBody=*/[{}], + /*defaultImplementation=*/[{ + return true; + }]>]; let verify = [{ return ::mlir::memref::detail::verifyIndexedMemCopyOpInterface($_op); }]; diff --git a/examples/mlir/Dialect/PDL/PDLDialect.td b/examples/mlir/Dialect/PDL/PDLDialect.td index d405bec..b989405 100644 --- a/examples/mlir/Dialect/PDL/PDLDialect.td +++ b/examples/mlir/Dialect/PDL/PDLDialect.td @@ -63,6 +63,7 @@ def PDL_Dialect : Dialect { }]; let name = "pdl"; + let useStrictPropertiesInAssemblyFormat = 1; let cppNamespace = "::mlir::pdl"; let useDefaultTypePrinterParser = 1; diff --git a/examples/mlir/Dialect/PDL/PDLOps.td b/examples/mlir/Dialect/PDL/PDLOps.td index 6ee638c..f2279d3 100644 --- a/examples/mlir/Dialect/PDL/PDLOps.td +++ b/examples/mlir/Dialect/PDL/PDLOps.td @@ -52,7 +52,8 @@ def PDL_ApplyNativeConstraintOp DefaultValuedAttr:$isNegated); let results = (outs Variadic:$results); let assemblyFormat = [{ - $name `(` $args `:` type($args) `)` (`:` type($results)^ )? attr-dict + $name `(` $args `:` type($args) `)` + (`is_negated` `=` $isNegated^)? (`:` type($results)^ )? attr-dict }]; let hasVerifier = 1; } @@ -392,7 +393,7 @@ def PDL_OperationOp : PDL_Op<"operation", [AttrSizedOperandSegments]> { //===----------------------------------------------------------------------===// def PDL_PatternOp : PDL_Op<"pattern", [ - IsolatedFromAbove, SingleBlock, Symbol, + IsolatedFromAbove, SingleBlock, SymbolName, SymbolVisibility, Symbol, DeclareOpInterfaceMethods ]> { let summary = "Define a rewrite pattern"; @@ -420,10 +421,12 @@ def PDL_PatternOp : PDL_Op<"pattern", [ }]; let arguments = (ins ConfinedAttr:$benefit, - OptionalAttr:$sym_name); + OptionalAttr:$sym_name, + OptionalAttr:$sym_visibility); let regions = (region SizedRegion<1>:$bodyRegion); let assemblyFormat = [{ - ($sym_name^)? `:` `benefit` `(` $benefit `)` attr-dict-with-keyword $bodyRegion + ($sym_visibility^)? ($sym_name^)? `:` `benefit` `(` $benefit `)` + attr-dict-with-keyword $bodyRegion }]; let builders = [ diff --git a/examples/mlir/Dialect/SCF/SCFOps.td b/examples/mlir/Dialect/SCF/SCFOps.td index 0b33ecb..b0a3498 100644 --- a/examples/mlir/Dialect/SCF/SCFOps.td +++ b/examples/mlir/Dialect/SCF/SCFOps.td @@ -27,6 +27,7 @@ include "mlir/Interfaces/ViewLikeInterface.td" def SCF_Dialect : Dialect { let name = "scf"; let cppNamespace = "::mlir::scf"; + let useStrictPropertiesInAssemblyFormat = 1; let description = [{ The `scf` (structured control flow) dialect contains operations that @@ -41,7 +42,9 @@ def SCF_Dialect : Dialect { and then lowered to some final target like LLVM or SPIR-V. }]; - let dependentDialects = ["arith::ArithDialect"]; + // The canonicalization of a multi-block `scf.execute_region` materializes + // `cf.br` operations, so the ControlFlow dialect must always be loaded. + let dependentDialects = ["arith::ArithDialect", "cf::ControlFlowDialect"]; } // Base class for SCF dialect ops. @@ -78,7 +81,8 @@ def ConditionOp : SCF_Op<"condition", [ //===----------------------------------------------------------------------===// def ExecuteRegionOp : SCF_Op<"execute_region", [ - DeclareOpInterfaceMethods, + DeclareOpInterfaceMethods, DeclareOpInterfaceMethods, RecursiveMemoryEffects]> { let summary = "operation that executes its region exactly once"; @@ -168,19 +172,27 @@ def ForOp : SCF_Op<"for", RecursiveMemoryEffects]> { let summary = "for operation"; let description = [{ - The `scf.for` operation represents a loop taking 3 SSA value as operands - that represent the lower bound, upper bound and step respectively. The - operation defines an SSA value for its induction variable. It has one - region capturing the loop body. The induction variable is represented as an - argument of this region. This SSA value is a signless integer or index. - The step is a value of same type but required to be positive, the lower and - upper bounds can be also negative or zero. The lower and upper bounds - specify a half-open range: the iteration is executed iff the comparison of - induction variable value is less than the upper bound and bigger or equal - to the lower bound. - - By default, the integer comparison is signed. If the `unsignedCmp` unit - attribute is specified, the integer comparison is unsigned. + The `scf.for` operation represents a loop whose first three operands are the + lower bound, upper bound and step respectively. The operation has one region + capturing the loop body. The induction variable is represented as an + argument of this region. + + Lower bound, upper bound, and step are interpreted as signed integers by + default, or as unsigned integers if the `unsignedCmp` unit attribute is + present. The step is required to be strictly positive. + + The lower and upper bounds specify a half-open range, including the lower + bound but excluding the upper bound. More precisely, the semantics is + governed by the following two rules, where arithmetic is performed with + arbitrary precision: + + 1. The trip count `n` is `max(0, ceil((UB - LB) / Step))` and the induction + variable takes the values `LB + j*Step` for `j = 0, ..., n - 1` in that + order. + 2. No-overflow condition: `LB + n*Step` must be representable in the type of + the induction variable. Otherwise the behavior is undefined. Leaving this + case undefined lets the loop be lowered to a plain increment-and-compare + on a single induction register without having to account for wraparound. The body region must contain exactly one block that terminates with `scf.yield`. Calling ForOp::build will create such a region and insert diff --git a/examples/mlir/Dialect/Tensor/TensorOps.td b/examples/mlir/Dialect/Tensor/TensorOps.td index c9b8585..beac1e9 100644 --- a/examples/mlir/Dialect/Tensor/TensorOps.td +++ b/examples/mlir/Dialect/Tensor/TensorOps.td @@ -402,8 +402,8 @@ def Tensor_ExtractSliceOp : Tensor_OpWithOffsetSizesAndStrides<"extract_slice", Note that there may be multiple ways to infer a resulting rank-reduced type. e.g. 1x6x1 could potentially rank-reduce to either 1x6 or 6x1 2-D shapes. - To disambiguate, the inference helpers `inferCanonicalRankReducedResultType` - only drop the first unit dimensions, in order: + To disambiguate, canonical rank-reduction inference drops only the first + unit dimensions, in order: e.g. 1x6x1 rank-reduced to 2-D will infer the 6x1 2-D shape, but not 1x6. Verification however has access to result type and does not need to infer. @@ -503,23 +503,6 @@ def Tensor_ExtractSliceOp : Tensor_OpWithOffsetSizesAndStrides<"extract_slice", RankedTensorType sourceTensorType, ArrayRef staticSizes); - /// If the rank is reduced (i.e. the desiredResultRank is smaller than the - /// number of sizes), drop as many size 1 as needed to produce an inferred type - /// with the desired rank. - /// - /// Note that there may be multiple ways to compute this rank-reduced type: - /// e.g. 1x6x1 can rank-reduce to either 1x6 or 6x1 2-D tensors. - /// - /// To disambiguate, this function always drops the first 1 sizes occurrences. - static RankedTensorType inferCanonicalRankReducedResultType( - unsigned resultRank, - RankedTensorType sourceRankedTensorType, - ArrayRef staticSizes); - static RankedTensorType inferCanonicalRankReducedResultType( - unsigned resultRank, - RankedTensorType sourceRankedTensorType, - ArrayRef staticSizes); - /// Return the expected rank of each of the`static_offsets`, `static_sizes` /// and `static_strides` attributes. std::array getArrayAttrMaxRanks() { @@ -1235,9 +1218,11 @@ def Tensor_CollapseShapeOp : Tensor_ReassociativeReshapeOp<"collapse_shape"> { "ArrayRef":$reassociation, CArg<"ArrayRef", "{}">:$attrs), [{ - $_state.addAttribute("reassociation", - getReassociationIndicesAttribute($_builder, reassociation)); - build($_builder, $_state, resultType, src, attrs); + buildPropertiesAndDiscardableAttributes($_state, attrs); + $_state.getOrAddProperties().reassociation = + getReassociationIndicesAttribute($_builder, reassociation); + $_state.addOperands(src); + $_state.addTypes(resultType); }]>, OpBuilder<(ins "Type":$resultType, "Value":$src, "ArrayRef":$reassociation, diff --git a/examples/mlir/Dialect/Transform/TransformDialect.td b/examples/mlir/Dialect/Transform/TransformDialect.td index ce0ad30..d537610 100644 --- a/examples/mlir/Dialect/Transform/TransformDialect.td +++ b/examples/mlir/Dialect/Transform/TransformDialect.td @@ -19,6 +19,7 @@ def Transform_Dialect : Dialect { let cppNamespace = "::mlir::transform"; let hasOperationAttrVerify = 1; + let useStrictPropertiesInAssemblyFormat = 1; let extraClassDeclaration = [{ /// Symbol name for the default entry point "named sequence". constexpr const static ::llvm::StringLiteral diff --git a/examples/mlir/Dialect/Transform/TransformOps.td b/examples/mlir/Dialect/Transform/TransformOps.td index 2daa2ad..ee5b6ae 100644 --- a/examples/mlir/Dialect/Transform/TransformOps.td +++ b/examples/mlir/Dialect/Transform/TransformOps.td @@ -222,8 +222,7 @@ def ApplyConversionPatternsOp : TransformDialectOp<"apply_conversion_patterns", let assemblyFormat = [{ `to` $target $patterns - (`with` `type_converter` $default_type_converter_region^)? - attr-dict `:` type($target) + (`with` `type_converter` $default_type_converter_region^)? prop-dict attr-dict `:` type($target) }]; let hasVerifier = 1; @@ -339,7 +338,7 @@ def ApplyPatternsOp : TransformDialectOp<"apply_patterns", let results = (outs); let regions = (region MaxSizedRegion<1>:$patterns); - let assemblyFormat = "`to` $target $patterns attr-dict `:` type($target)"; + let assemblyFormat = "`to` $target $patterns oilist(`apply_cse` $apply_cse | `max_iterations` `=` $max_iterations | `max_num_rewrites` `=` $max_num_rewrites) attr-dict `:` type($target)"; let hasVerifier = 1; let skipDefaultBuilders = 1; @@ -613,8 +612,7 @@ def ForeachMatchOp : TransformDialectOp<"foreach_match", [ ) `in` $root (`,` $forwarded_inputs^)? - custom($matchers, $actions) - attr-dict + custom($matchers, $actions) attr-dict `:` functional-type(operands, results) }]; @@ -759,7 +757,7 @@ def GetParentOp : TransformDialectOp<"get_parent_op", "1">:$nth_parent); let results = (outs TransformHandleTypeInterface:$parent); let assemblyFormat = - "$target attr-dict `:` functional-type(operands, results)"; + "$target prop-dict attr-dict `:` functional-type(operands, results)"; } def GetProducerOfOperand : TransformDialectOp<"get_producer_of_operand", @@ -907,7 +905,7 @@ def IncludeOp : TransformDialectOp<"include", let assemblyFormat = "$target `failures` `(` $failure_propagation_mode `)`" - "`(` $operands `)` attr-dict `:` functional-type($operands, $results)"; + "`(` $operands `)` oilist(`arg_attrs` `=` $arg_attrs | `res_attrs` `=` $res_attrs) attr-dict `:` functional-type($operands, $results)"; let extraClassDeclaration = [{ ::mlir::CallInterfaceCallable getCallableForCallee() { @@ -1007,7 +1005,7 @@ def MergeHandlesOp : TransformDialectOp<"merge_handles", } def NamedSequenceOp : TransformDialectOp<"named_sequence", - [FunctionOpInterface, + [SymbolName, SymbolVisibility, FunctionOpInterface, IsolatedFromAbove, DeclareOpInterfaceMethods, DeclareOpInterfaceMethods]> { @@ -1116,7 +1114,7 @@ def SplitHandleOp : TransformDialectOp<"split_handle", ]; let assemblyFormat = [{ - $handle attr-dict `:` functional-type(operands, results) + $handle oilist(`pass_through_empty_handle` `=` $pass_through_empty_handle | `fail_on_payload_too_small` `=` $fail_on_payload_too_small | `overflow_result` `=` $overflow_result) attr-dict `:` functional-type(operands, results) }]; } @@ -1152,7 +1150,7 @@ def PayloadOp : Op:$normal_forms); - let assemblyFormat = "attr-dict-with-keyword regions"; + let assemblyFormat = "`normal_forms` `=` $normal_forms attr-dict-with-keyword regions"; } def PrintOp : TransformDialectOp<"print", @@ -1190,7 +1188,7 @@ def PrintOp : TransformDialectOp<"print", OpBuilder<(ins "Value":$target, CArg<"StringRef", "StringRef()">:$name)> ]; - let assemblyFormat = "$target attr-dict (`:` type($target)^)?"; + let assemblyFormat = "$target oilist(`name` `=` $name | `assume_verified` $assume_verified | `use_local_scope` $use_local_scope | `skip_regions` $skip_regions) attr-dict (`:` type($target)^)?"; } def ReplicateOp : TransformDialectOp<"replicate", diff --git a/examples/mlir/Dialect/Vector/VectorAttributes.td b/examples/mlir/Dialect/Vector/VectorAttributes.td index bcf53da..58e18a1 100644 --- a/examples/mlir/Dialect/Vector/VectorAttributes.td +++ b/examples/mlir/Dialect/Vector/VectorAttributes.td @@ -45,9 +45,7 @@ def CombiningKind : I32EnumAttr< /// An attribute that specifies the combining function for `vector.contract`, /// and `vector.reduction`. -def Vector_CombiningKindAttr : EnumAttr { - let assemblyFormat = "`<` $value `>`"; -} +def Vector_CombiningKindAttr : EnumAttr; def Vector_IteratorType : I32EnumAttr<"IteratorType", "Iterator type", [ I32EnumAttrCase<"parallel", 0>, @@ -58,9 +56,7 @@ def Vector_IteratorType : I32EnumAttr<"IteratorType", "Iterator type", [ } def Vector_IteratorTypeEnum - : EnumAttr { - let assemblyFormat = "`<` $value `>`"; -} + : EnumAttr; def Vector_IteratorTypeArrayAttr : TypedArrayAttrBase { - let assemblyFormat = "`<` $value `>`"; -} +def Vector_PrintPunctuation : EnumAttr; #endif // MLIR_DIALECT_VECTOR_IR_VECTOR_ATTRIBUTES diff --git a/examples/mlir/Dialect/Vector/VectorOps.td b/examples/mlir/Dialect/Vector/VectorOps.td index 8f4fa5c..3309e8a 100644 --- a/examples/mlir/Dialect/Vector/VectorOps.td +++ b/examples/mlir/Dialect/Vector/VectorOps.td @@ -584,6 +584,8 @@ def Vector_InterleaveOp : return ::llvm::cast(getResult().getType()); } }]; + let hasCanonicalizer = 1; + let hasFolder = 1; } class ResultIsHalfSourceVectorType : TypesMatchWith< @@ -1073,13 +1075,15 @@ def Vector_InsertStridedSliceOp : ```mlir %2 = vector.insert_strided_slice %0, %1 - {offsets = [0, 0, 2], strides = [1, 1]}: + offsets = [0, 0, 2], strides = [1, 1] : vector<2x4xf32> into vector<16x4x8xf32> ``` }]; let assemblyFormat = [{ - $valueToStore `,` $dest attr-dict `:` type($valueToStore) `into` type($dest) + $valueToStore `,` $dest + `offsets` `=` $offsets `,` `strides` `=` $strides attr-dict + `:` type($valueToStore) `into` type($dest) }]; let builders = [ @@ -1215,7 +1219,7 @@ def Vector_ExtractStridedSliceOp : ```mlir %1 = vector.extract_strided_slice %0 - {offsets = [0, 2], sizes = [2, 4], strides = [1, 1]}: + offsets = [0, 2], sizes = [2, 4], strides = [1, 1] : vector<4x8x16xf32> to vector<2x4x16xf32> // TODO: Evolve to a range form syntax similar to: @@ -1243,7 +1247,10 @@ def Vector_ExtractStridedSliceOp : let hasCanonicalizer = 1; let hasFolder = 1; let hasVerifier = 1; - let assemblyFormat = "$source attr-dict `:` type($source) `to` type(results)"; + let assemblyFormat = [{ + $source `offsets` `=` $offsets `,` `sizes` `=` $sizes `,` + `strides` `=` $strides attr-dict `:` type($source) `to` type(results) + }]; } // TODO: Tighten semantics so that masks and inbounds can't be used @@ -1456,6 +1463,7 @@ def Vector_TransferReadOp : let builders = [ /// 1. Builder that sets padding to `padding` or poison if not provided and /// an empty mask (variant with attrs). + /// If `padding` is null, a poison value is used. OpBuilder<(ins "VectorType":$vectorType, "Value":$source, "ValueRange":$indices, @@ -1463,7 +1471,10 @@ def Vector_TransferReadOp : "AffineMapAttr":$permutationMapAttr, "ArrayAttr":$inBoundsAttr)>, /// 2. Builder that sets padding to `padding` or poison if not provided and - /// an empty mask (variant without attrs). + /// an empty mask (variant without attrs). + /// If `padding` is null, a poison value is used. + /// If `permutationMap` is null, a minor identity map is used. + /// If `inBounds` is null, an empty mask is used. OpBuilder<(ins "VectorType":$vectorType, "Value":$source, "ValueRange":$indices, @@ -1472,6 +1483,8 @@ def Vector_TransferReadOp : CArg<"std::optional>", "::std::nullopt">:$inBounds)>, /// 3. Builder that sets padding to `padding` or poison if not provided and /// permutation map to 'getMinorIdentityMap'. + /// If `padding` is null, a poison value is used. + /// If `inBounds` is null, an empty mask is used. OpBuilder<(ins "VectorType":$vectorType, "Value":$source, "ValueRange":$indices, @@ -1630,7 +1643,8 @@ def Vector_TransferWriteOp : "ValueRange":$indices, "AffineMapAttr":$permutationMapAttr, "ArrayAttr":$inBoundsAttr)>, - /// 3. Builder with type inference that sets an empty mask (variant without attrs). + /// 3. Builder with type inference that sets an empty mask (variant without + /// attrs). If `permutationMap` is null, a minor identity map is used. OpBuilder<(ins "Value":$vector, "Value":$dest, "ValueRange":$indices, @@ -1658,6 +1672,7 @@ def Vector_TransferWriteOp : let hasVerifier = 1; } +// Promises IndexedAccessOpInterface. def Vector_LoadOp : Vector_Op<"load", [ DeclareOpInterfaceMethods, DeclareOpInterfaceMethods, @@ -1708,6 +1723,9 @@ def Vector_LoadOp : Vector_Op<"load", [ %result = vector.load %memref[%i, %j] : memref<200x100xvector<4x8xf32>>, vector<4x8xf32> ``` + The memref must have non-negative strides. Negative strides are not supported + and will trigger a verification error. + Representation-wise, the 'vector.load' operation permits out-of-bounds reads. Support and implementation of out-of-bounds vector loads is target-specific. No assumptions should be made on the value of elements @@ -1772,10 +1790,17 @@ def Vector_LoadOp : Vector_Op<"load", [ let hasFolder = 1; let hasVerifier = 1; - let assemblyFormat = - "$base `[` $indices `]` attr-dict `:` type($base) `,` type($result)"; + let assemblyFormat = [{ + $base `[` $indices `]` + oilist( + `alignment` `=` $alignment | + `nontemporal` `=` custom($nontemporal) + ) + attr-dict `:` type($base) `,` type($result) + }]; } +// Promises IndexedAccessOpInterface. def Vector_StoreOp : Vector_Op<"store", [ DeclareOpInterfaceMethods, DeclareOpInterfaceMethods, @@ -1825,6 +1850,9 @@ def Vector_StoreOp : Vector_Op<"store", [ vector.store %valueToStore, %memref[%i, %j] : memref<200x100xvector<4x8xf32>>, vector<4x8xf32> ``` + The memref must have non-negative strides. Negative strides are not supported + and will trigger a verification error. + Representation-wise, the 'vector.store' operation permits out-of-bounds writes. Support and implementation of out-of-bounds vector stores are target-specific. No assumptions should be made on the memory written out of @@ -1879,10 +1907,17 @@ def Vector_StoreOp : Vector_Op<"store", [ let hasFolder = 1; let hasVerifier = 1; - let assemblyFormat = "$valueToStore `,` $base `[` $indices `]` attr-dict " - "`:` type($base) `,` type($valueToStore)"; + let assemblyFormat = [{ + $valueToStore `,` $base `[` $indices `]` + oilist( + `alignment` `=` $alignment | + `nontemporal` `=` custom($nontemporal) + ) + attr-dict `:` type($base) `,` type($valueToStore) + }]; } +// Promises IndexedAccessOpInterface. def Vector_MaskedLoadOp : Vector_Op<"maskedload", [ DeclareOpInterfaceMethods, @@ -1929,6 +1964,9 @@ def Vector_MaskedLoadOp : : memref, vector<16xi1>, vector<16xf32> into vector<16xf32> ``` + The memref must have non-negative strides. Negative strides are not supported + and will trigger a verification error. + An optional `alignment` attribute allows to specify the byte alignment of the load operation. It must be a positive power of 2. The operation must access memory at an address aligned to this boundary. Violating this requirement @@ -1948,8 +1986,12 @@ def Vector_MaskedLoadOp : return ::llvm::cast(getResult().getType()); } }]; - let assemblyFormat = "$base `[` $indices `]` `,` $mask `,` $pass_thru attr-dict `:` " - "type($base) `,` type($mask) `,` type($pass_thru) `into` type($result)"; + let assemblyFormat = [{ + $base `[` $indices `]` `,` $mask `,` $pass_thru + (`alignment` `=` $alignment^)? + attr-dict `:` type($base) `,` type($mask) `,` + type($pass_thru) `into` type($result) + }]; let hasCanonicalizer = 1; let hasFolder = 1; let hasVerifier = 1; @@ -1978,6 +2020,7 @@ def Vector_MaskedLoadOp : ]; } +// Promises IndexedAccessOpInterface. def Vector_MaskedStoreOp : Vector_Op<"maskedstore", [ DeclareOpInterfaceMethods, @@ -2023,6 +2066,9 @@ def Vector_MaskedStoreOp : : memref, vector<16xi1>, vector<16xf32> ``` + The memref must have non-negative strides. Negative strides are not supported + and will trigger a verification error. + An optional `alignment` attribute allows to specify the byte alignment of the store operation. It must be a positive power of 2. The operation must access memory at an address aligned to this boundary. Violating this requirement @@ -2039,9 +2085,11 @@ def Vector_MaskedStoreOp : return ::llvm::cast(getValueToStore().getType()); } }]; - let assemblyFormat = - "$base `[` $indices `]` `,` $mask `,` $valueToStore " - "attr-dict `:` type($base) `,` type($mask) `,` type($valueToStore)"; + let assemblyFormat = [{ + $base `[` $indices `]` `,` $mask `,` $valueToStore + (`alignment` `=` $alignment^)? + attr-dict `:` type($base) `,` type($mask) `,` type($valueToStore) + }]; let hasCanonicalizer = 1; let hasFolder = 1; let hasVerifier = 1; @@ -2117,6 +2165,9 @@ def Vector_GatherOp : during progressively lowering to bring other memory operations closer to hardware ISA support for a gather. + The memref must have non-negative strides. Negative strides are not supported + and will trigger a verification error. + An optional `alignment` attribute allows to specify the byte alignment of the gather operation. It must be a positive power of 2. The operation must access memory at an address aligned to this boundary. Violating this requirement @@ -2145,7 +2196,8 @@ def Vector_GatherOp : let assemblyFormat = "$base `[` $offsets `]` `[` $indices `]` `,` " - "$mask `,` $pass_thru attr-dict `:` type($base) `,` " + "$mask `,` $pass_thru (`alignment` `=` $alignment^)? " + "attr-dict `:` type($base) `,` " "type($indices) `,` type($mask) `,` type($pass_thru) " "`into` type($result)"; let hasCanonicalizer = 1; @@ -2210,6 +2262,9 @@ def Vector_ScatterOp correspond to those of the `llvm.masked.scatter` [intrinsic](https://llvm.org/docs/LangRef.html#llvm-masked-scatter-intrinsics). + The memref must have non-negative strides. Negative strides are not supported + and will trigger a verification error. + An optional `alignment` attribute allows to specify the byte alignment of the scatter operation. It must be a positive power of 2. The operation must access memory at an address aligned to this boundary. Violating this requirement @@ -2233,10 +2288,12 @@ def Vector_ScatterOp VectorType getVectorType() { return getValueToStore().getType(); } }]; - let assemblyFormat = "$base `[` $offsets `]` `[` $indices `]` `,` " - "$mask `,` $valueToStore attr-dict `:` type($base) `,` " - "type($indices) `,` type($mask) `,` " - "type($valueToStore) (`->` type($result)^)?"; + let assemblyFormat = [{ + $base `[` $offsets `]` `[` $indices `]` `,` $mask `,` $valueToStore + (`alignment` `=` $alignment^)? + attr-dict `:` type($base) `,` type($indices) `,` type($mask) `,` + type($valueToStore) (`->` type($result)^)? + }]; let hasCanonicalizer = 1; let hasVerifier = 1; @@ -2251,6 +2308,7 @@ def Vector_ScatterOp }]>]; } +// Promises IndexedAccessOpInterface. def Vector_ExpandLoadOp : Vector_Op<"expandload", [ DeclareOpInterfaceMethods, @@ -2258,7 +2316,7 @@ def Vector_ExpandLoadOp : ]>, Arguments<(ins Arg:$base, Variadic:$indices, - FixedVectorOfNonZeroRankOf<[I1]>:$mask, + VectorOfNonZeroRankOf<[I1]>:$mask, AnyVectorOfNonZeroRank:$pass_thru, OptionalAttr>: $alignment)>, Results<(outs AnyVectorOfNonZeroRank:$result)> { @@ -2297,8 +2355,6 @@ def Vector_ExpandLoadOp : memory at an address aligned to this boundary. Violating this requirement triggers immediate undefined behavior. - Note, at the moment this Op is only available for fixed-width vectors. - Examples: ```mlir @@ -2323,8 +2379,12 @@ def Vector_ExpandLoadOp : return ::llvm::cast(getResult().getType()); } }]; - let assemblyFormat = "$base `[` $indices `]` `,` $mask `,` $pass_thru attr-dict `:` " - "type($base) `,` type($mask) `,` type($pass_thru) `into` type($result)"; + let assemblyFormat = [{ + $base `[` $indices `]` `,` $mask `,` $pass_thru + (`alignment` `=` $alignment^)? + attr-dict `:` type($base) `,` type($mask) `,` + type($pass_thru) `into` type($result) + }]; let hasCanonicalizer = 1; let hasVerifier = 1; @@ -2342,6 +2402,7 @@ def Vector_ExpandLoadOp : ]; } +// Promises IndexedAccessOpInterface. def Vector_CompressStoreOp : Vector_Op<"compressstore", [ DeclareOpInterfaceMethods, @@ -2349,7 +2410,7 @@ def Vector_CompressStoreOp : ]>, Arguments<(ins Arg:$base, Variadic:$indices, - FixedVectorOfNonZeroRankOf<[I1]>:$mask, + VectorOfNonZeroRankOf<[I1]>:$mask, AnyVectorOfNonZeroRank:$valueToStore, OptionalAttr>: $alignment)> { @@ -2387,8 +2448,6 @@ def Vector_CompressStoreOp : memory at an address aligned to this boundary. Violating this requirement triggers immediate undefined behavior. - Note, at the moment this Op is only available for fixed-width vectors. - Examples: ```mlir @@ -2410,9 +2469,11 @@ def Vector_CompressStoreOp : return ::llvm::cast(getValueToStore().getType()); } }]; - let assemblyFormat = - "$base `[` $indices `]` `,` $mask `,` $valueToStore attr-dict `:` " - "type($base) `,` type($mask) `,` type($valueToStore)"; + let assemblyFormat = [{ + $base `[` $indices `]` `,` $mask `,` $valueToStore + (`alignment` `=` $alignment^)? + attr-dict `:` type($base) `,` type($mask) `,` type($valueToStore) + }]; let hasCanonicalizer = 1; let hasVerifier = 1; let builders = [ @@ -2464,6 +2525,9 @@ def Vector_ShapeCastOp : VectorType getResultVectorType() { return ::llvm::cast(getResult().getType()); } + // Return true if this Op is effectively a vector.broadcast (i.e. the input + // and output shapes satisfy the vector.broadcast constraints). + bool isBroadcastLike(); }]; let assemblyFormat = "$source attr-dict `:` type($source) `to` type($result)"; let hasFolder = 1; @@ -2554,7 +2618,7 @@ def Vector_TypeCastOp : } def Vector_ConstantMaskOp : - Vector_Op<"constant_mask", [Pure, + Vector_Op<"constant_mask", [Pure, DeclareOpInterfaceMethods ]>, Arguments<(ins DenseI64ArrayAttr:$mask_dim_sizes)>, @@ -2614,7 +2678,7 @@ def Vector_ConstantMaskOp : } def Vector_CreateMaskOp : - Vector_Op<"create_mask", [Pure, + Vector_Op<"create_mask", [Pure, DeclareOpInterfaceMethods ]>, Arguments<(ins Variadic:$mask_dim_sizes)>, @@ -2976,7 +3040,7 @@ def Vector_ScanOp : Example: ```mlir - %1:2 = vector.scan , %0, %acc {inclusive = false, reduction_dim = 1 : i64} : + %1:2 = vector.scan , %0, %acc reduction_dim = 1, inclusive = false : vector<4x8x16x32xf32>, vector<4x16x32xf32> ``` }]; @@ -2995,9 +3059,11 @@ def Vector_ScanOp : return ::llvm::cast(getInitialValue().getType()); } }]; - let assemblyFormat = - "$kind `,` $source `,` $initial_value attr-dict `:` " - "type($source) `,` type($initial_value) "; + let assemblyFormat = [{ + $kind `,` $source `,` $initial_value + `reduction_dim` `=` $reduction_dim `,` `inclusive` `=` $inclusive + attr-dict `:` type($source) `,` type($initial_value) + }]; let hasVerifier = 1; } @@ -3005,6 +3071,10 @@ def Vector_ScanOp : // VectorStepOp //===----------------------------------------------------------------------===// +def VectorStepElementType : Type< + CPred<"::llvm::isa<::mlir::IndexType>($_self) || ($_self.isSignlessInteger() && $_self.getIntOrFloatBitWidth() >= 8)">, + "index or signless integer of at least 8 bits">; + def Vector_StepOp : Vector_Op<"step", [ Pure, DeclareOpInterfaceMethods, @@ -3012,20 +3082,25 @@ def Vector_StepOp : Vector_Op<"step", [ ]> { let summary = "A linear sequence of values from 0 to N"; let description = [{ - A `step` operation produces an index vector, i.e. a 1-D vector of values of - index type that represents a linear sequence from 0 to N-1, where N is the - number of elements in the `result` vector. + A `step` operation produces a 1-D vector representing a linear sequence from + 0 to N-1, where N is the number of elements in the `result` vector. + + The result element type must be `index` or a signless integer of at least 8 + bits. If the sequence value exceeds the allowed limit for the element type + then the result for that lane is truncated. Supports fixed-width and scalable vectors. Examples: ```mlir - %0 = vector.step : vector<4xindex> ; [0, 1, 2, 3] - %1 = vector.step : vector<[4]xindex> ; [0, 1, .., ] + %0 = vector.step : vector<4xindex> // [0, 1, 2, 3] + %1 = vector.step : vector<4xi32> // [0, 1, 2, 3] + %2 = vector.step : vector<258xi8> // [0, 1, .., 255, 0, 1] + %3 = vector.step : vector<[4]xindex> // [0, 1, .., ] ``` }]; - let results = (outs VectorOfRankAndType<[1], [Index]>:$result); + let results = (outs VectorOfRankAndType<[1], [VectorStepElementType]>:$result); let assemblyFormat = "attr-dict `:` type($result)"; let hasCanonicalizer = 1; } diff --git a/examples/mlir/IR/BuiltinAttributes.td b/examples/mlir/IR/BuiltinAttributes.td index 6165a24..ba2c374 100644 --- a/examples/mlir/IR/BuiltinAttributes.td +++ b/examples/mlir/IR/BuiltinAttributes.td @@ -160,6 +160,9 @@ def Builtin_DenseArrayRawDataParameter : ArrayRefParameter< $_allocator.allocate($_self.size(), alignof(uint64_t))); llvm::uninitialized_copy($_self, alloc); $_dst = ArrayRef(alloc, $_self.size()); + } else { + // Align possibly unaligned empty range. + $_dst = {}; } }]; } @@ -271,7 +274,7 @@ def Builtin_DenseTypedElementsAttr : Builtin_Attr< dense : 10 : i32> // Type-first syntax: A tensor of 2 float32 elements. - dense : [10.0, 11.0]> + dense : [10.0 : f32, 11.0 : f32]> ``` Note: The literal-first syntax is supported only for complex, float, index, diff --git a/examples/mlir/IR/BuiltinDialectBytecode.td b/examples/mlir/IR/BuiltinDialectBytecode.td index c97d093..d098793 100644 --- a/examples/mlir/IR/BuiltinDialectBytecode.td +++ b/examples/mlir/IR/BuiltinDialectBytecode.td @@ -205,6 +205,50 @@ def DistinctAttr : DialectAttribute<(attr Attribute:$referencedAttr )>; +// Make easy to disable until version number gets added. +class EnableAffineMapPrintingJuly2026 : DialectAttribute; + +def AffineMapAttr : EnableAffineMapPrintingJuly2026<(attr + WithParser<"succeeded(readAffineMap($_reader, context, $_var))", + WithPrinter<"writeAffineMap($_writer, $_name)", + WithType<"AffineMap">>>:$value +)>; + +def AffineExpr : + WithParser<"succeeded(readAffineExpr($_reader, context, $_var))", + WithBuilder<"$_args", + WithPrinter<"writeAffineExpr($_writer, $_getter)", + WithType<"AffineExpr">>>>; +def AffineExprList : List; + +// Similar to AffineMapAttr, IntegerSetAttr has everything going via IntegerSet +// and getValue, so extra indirections. +def IntegerSetConstraints : Array { + let cGetter = "$_attrType.getValue().getConstraints()"; +} + +def IntegerSetEqFlags : ArrayWithKnownSize { + let cGetter = "$_attrType.getValue().getEqFlags()"; +} + +class EnableIntegerSetPrintingJuly2026 : DialectType; + +def IntegerSetAttr : EnableIntegerSetPrintingJuly2026<(attr + WithGetter<"$_attrType.getValue().getNumDims()", VarInt>:$numDims, + WithGetter<"$_attrType.getValue().getNumSymbols()", VarInt>:$numSymbols, + IntegerSetConstraints:$constraints, + IntegerSetEqFlags:$eqFlags +)> { + let cBuilder = "IntegerSetAttr::get(IntegerSet::get(numDims, numSymbols, constraints, eqFlags))"; +} + +class EnableStridedPrintingJuly2026 : DialectAttribute; + +def StridedLayoutAttr : EnableStridedPrintingJuly2026<(attr + SignedVarInt:$offset, + Array:$strides +)>; + // Types // ----- @@ -227,10 +271,23 @@ def FunctionType : DialectType<(type Array:$results )>; +def GraphType : DialectType<(type + Array:$inputs, + Array:$results +)>; + def BFloat16Type : DialectType<(type)>; def Float16Type : DialectType<(type)>; +// Make it easy to stage the addition of new floating point types so that +// readers can be updated first. This is enabled by default, but can be flipped +// to DialectTypeNoPrint if staging needed. +// Note: this will be removed post next release. +class EnableFloatPrintingJune2026 : DialectType; + +def FloatTF32Type : EnableFloatPrintingJune2026<(type)>; + def Float32Type : DialectType<(type)>; def Float64Type : DialectType<(type)>; @@ -239,6 +296,28 @@ def Float80Type : DialectType<(type)>; def Float128Type : DialectType<(type)>; +def Float8E5M2Type : EnableFloatPrintingJune2026<(type)>; + +def Float8E4M3Type : EnableFloatPrintingJune2026<(type)>; + +def Float8E4M3FNType : EnableFloatPrintingJune2026<(type)>; + +def Float8E5M2FNUZType : EnableFloatPrintingJune2026<(type)>; + +def Float8E4M3FNUZType : EnableFloatPrintingJune2026<(type)>; + +def Float8E4M3B11FNUZType : EnableFloatPrintingJune2026<(type)>; + +def Float8E3M4Type : EnableFloatPrintingJune2026<(type)>; + +def Float4E2M1FNType : EnableFloatPrintingJune2026<(type)>; + +def Float6E2M3FNType : EnableFloatPrintingJune2026<(type)>; + +def Float6E3M2FNType : EnableFloatPrintingJune2026<(type)>; + +def Float8E8M0FNUType : EnableFloatPrintingJune2026<(type)>; + def ComplexType : DialectType<(type Type:$elementType )>; @@ -294,6 +373,8 @@ def UnrankedTensorType : DialectType<(type Type:$elementType )>; +def TokenType : DialectType<(type)>; + let cType = "VectorType" in { def VectorType : DialectType<(type Array:$shape, @@ -346,6 +427,9 @@ def BuiltinDialectAttributes : DialectAttributes<"Builtin"> { SparseElementsAttr, DistinctAttr, FileLineColRange, + AffineMapAttr, + IntegerSetAttr, + StridedLayoutAttr, ]; } @@ -371,7 +455,21 @@ def BuiltinDialectTypes : DialectTypes<"Builtin"> { UnrankedMemRefTypeWithMemSpace, UnrankedTensorType, VectorType, - VectorTypeWithScalableDims + VectorTypeWithScalableDims, + FloatTF32Type, + Float8E5M2Type, + Float8E4M3Type, + Float8E4M3FNType, + Float8E5M2FNUZType, + Float8E4M3FNUZType, + Float8E4M3B11FNUZType, + Float8E3M4Type, + Float4E2M1FNType, + Float6E2M3FNType, + Float6E3M2FNType, + Float8E8M0FNUType, + TokenType, + GraphType ]; } diff --git a/examples/mlir/IR/BuiltinOps.td b/examples/mlir/IR/BuiltinOps.td index cdc09af..f16b686 100644 --- a/examples/mlir/IR/BuiltinOps.td +++ b/examples/mlir/IR/BuiltinOps.td @@ -31,7 +31,8 @@ class Builtin_Op traits = []> : //===----------------------------------------------------------------------===// def ModuleOp : Builtin_Op<"module", [ - AffineScope, IsolatedFromAbove, NoRegionArguments, SymbolTable, Symbol, + AffineScope, IsolatedFromAbove, NoRegionArguments, SymbolTable, SymbolName, + SymbolVisibility, Symbol, OpAsmOpInterface ] # GraphRegionNoTerminator.traits> { let summary = "A top level container operation"; diff --git a/examples/mlir/IR/BuiltinTypes.td b/examples/mlir/IR/BuiltinTypes.td index 20c41c5..ffb0e54 100644 --- a/examples/mlir/IR/BuiltinTypes.td +++ b/examples/mlir/IR/BuiltinTypes.td @@ -433,6 +433,26 @@ def Builtin_Float8E8M0FNU : Builtin_FloatType<"Float8E8M0FNU", "f8E8M0FNU"> { }]; } +//===----------------------------------------------------------------------===// +// Float8E5M3FNUType +//===----------------------------------------------------------------------===// + +def Builtin_Float8E5M3FNU : Builtin_FloatType<"Float8E5M3FNU", "f8E5M3FNU"> { + let summary = "8-bit unsigned floating point with 5-bit exponent, 3-bit mantissa"; + let description = [{ + An 8-bit floating point type with 0 sign bit, 5 bits exponent and 3 bits + mantissa. This is not a standard type as defined by IEEE-754, but it + follows similar conventions with the following characteristics: + + * bit encoding: S0E5M3 + * exponent bias: 15 + * infinities: Not supported + * NaNs: Supported with all bits set to 1 + * Zero: Supported + * denormals when exponent is 0 + }]; +} + //===----------------------------------------------------------------------===// // BFloat16Type //===----------------------------------------------------------------------===// @@ -874,7 +894,8 @@ def Builtin_MemRef : Builtin_Type<"MemRef", "memref", [ form which is converted to a semi-affine map automatically. The memory space of a memref is specified by a target-specific attribute. - It might be an integer value, string, dictionary or custom dialect attribute. + It might be an integer value, string, dictionary or custom dialect + attribute; no restriction is placed on the kind of attribute used. The empty memory space (attribute is None) is target specific. The notionally dynamic value of a memref value includes the address of the @@ -1237,6 +1258,28 @@ def Builtin_RankedTensor : Builtin_Type<"RankedTensor", "tensor", [ let genVerifyDecl = 1; } +//===----------------------------------------------------------------------===// +// TokenType +//===----------------------------------------------------------------------===// + +def Builtin_Token : Builtin_Type<"Token", "token"> { + let summary = "Token type"; + let description = [{ + Syntax: + + ``` + token-type ::= `token` + ``` + + A use of a token SSA value is a pointer to an operation (in case of an + OpResult) or a pointer to a region (in case of an entry block argument). + A token carries no runtime data and cannot be forwarded. Tokens are + excluded from the `AnyType` type constraint. Operations must define + `TokenProducerTrait` to produce token results or token region entry block + arguments, and must define `TokenConsumerTrait` to consume token operands. + }]; +} + //===----------------------------------------------------------------------===// // TupleType //===----------------------------------------------------------------------===// diff --git a/examples/mlir/IR/BytecodeBase.td b/examples/mlir/IR/BytecodeBase.td index 184c81e..3eab624 100644 --- a/examples/mlir/IR/BytecodeBase.td +++ b/examples/mlir/IR/BytecodeBase.td @@ -125,6 +125,15 @@ class Array { Bytecode elemT = t; string cBuilder = "$_args"; + + // Optional custom getter expression for the array from the parent type. + string cGetter = ""; +} +// - Array variant where the size is known from a previously serialized member. +// This avoids encoding a redundant length prefix. +class ArrayWithKnownSize : Array { + // Expression providing the element count (e.g., "otherMember.size()"). + string knownSizeRef = sizeRef; } // - Array elements currently needs a different bytecode type to accommodate // for the list print/parsing. @@ -153,6 +162,10 @@ class DialectType : DialectAttrOrType, TypeKind { let cParser = "succeeded($_reader.readType<$_resultType>($_var))"; let cBuilder = "getChecked<$_resultType>([&]() { return reader.emitError(); }, context, $_args)"; } +// Variant of the above, where it never prints. Useful for staging. +class DialectTypeNoPrint : DialectType { + let printerPredicate = "false"; +} class DialectAttributes { string dialect = d; @@ -173,4 +186,3 @@ def none; def ReservedOrDead : DialectAttrOrType<(none)>; #endif // BYTECODE_BASE - diff --git a/examples/mlir/IR/CommonTypeConstraints.td b/examples/mlir/IR/CommonTypeConstraints.td index 57caaae..8dbd7ac 100644 --- a/examples/mlir/IR/CommonTypeConstraints.td +++ b/examples/mlir/IR/CommonTypeConstraints.td @@ -127,7 +127,6 @@ class Variadic : TypeConstraint { Type baseType = type; - int minSize = 0; } // A nested variadic type constraint. It expands to zero or more variadic ranges @@ -165,8 +164,17 @@ class SameBuildabilityAs { code builderCall = !if(!empty(type.builderCall), "", builder); } -// Any type at all. -def AnyType : Type, "any type">; +// Whether a type is the builtin `TokenType`. +def IsTokenTypePred : CPred<"::llvm::isa<::mlir::TokenType>($_self)">; + +// Any non-token type. Tokens are excluded by default to prevent ops that +// accept arbitrary types from accidentally accepting tokens as operands / +// results, since a token must not be value-forwarded. +def AnyType : Type, "any non-token type">; + +// The builtin token type. +def Token : Type, + BuildableType<"$_builder.getType<::mlir::TokenType>()">; // None type def NoneType : Type($_self)">, "none type", @@ -369,6 +377,8 @@ def F6E3M2FN : Type($_self)">, "f6E BuildableType<"$_builder.getType()">; def F8E8M0FNU : Type($_self)">, "f8E8M0FNU type">, BuildableType<"$_builder.getType()">; +def F8E5M3FNU : Type($_self)">, "f8E5M3FNU type">, + BuildableType<"$_builder.getType()">; def AnyComplex : Type($_self)">, "complex-type", "::mlir::ComplexType">; @@ -416,13 +426,14 @@ class ContainerType allowedTypes, Pred containerPred, string descr, - string cppType = "::mlir::Type"> : + string cppType = "::mlir::Type", + string summary = ""> : Type.predicate>, "; }(::llvm::cast<::mlir::ShapedType>($_self).getElementType())">]>, - descr # " of " # AnyTypeOf.summary # " values", cppType>; + descr # " of " # AnyTypeOf.summary # " values" # summary, cppType>; // Whether a shaped type is ranked. def HasRankPred : CPred<"::llvm::cast<::mlir::ShapedType>($_self).hasRank()">; @@ -921,8 +932,8 @@ class AnyStridedMemRefOfRank : AnyStridedMemRef.summary # " of rank " # rank>; class StridedMemRefRankOf allowedTypes, list ranks> : - ConfinedType, [HasAnyRankOfPred], - !interleave(!foreach(rank, ranks, rank # "D"), "/") # " " # + ConfinedType, [HasAnyRankOfPred, HasStridesPred], + !interleave(!foreach(rank, ranks, rank # "D"), "/") # " strided " # MemRefOf.summary>; // This represents a generic tuple without any constraints on element type. diff --git a/examples/mlir/IR/DialectBase.td b/examples/mlir/IR/DialectBase.td index efa09a4..3b41e84 100644 --- a/examples/mlir/IR/DialectBase.td +++ b/examples/mlir/IR/DialectBase.td @@ -55,6 +55,12 @@ class Dialect { // dialect declaration. code extraClassDeclaration = ""; + // If this dialect should require declarative parsers for property-backed + // operations to bind every inherent attribute and property directly in the + // custom assembly format, or otherwise cover them with `prop-dict`. This + // stricter mode is disabled by default for now. + bit useStrictPropertiesInAssemblyFormat = 0; + // If this dialect overrides the hook for materializing constants. bit hasConstantMaterializer = 0; diff --git a/examples/mlir/IR/EnumAttr.td b/examples/mlir/IR/EnumAttr.td index 6eef507..1c3a223 100644 --- a/examples/mlir/IR/EnumAttr.td +++ b/examples/mlir/IR/EnumAttr.td @@ -503,6 +503,7 @@ class I64BitEnumAttr : AttrParameter { + EnumInfo enum = enumInfo; let parser = !if(!isa(enumInfo), !cast(enumInfo).parameterParser, ?); let printer = !if(!isa(enumInfo), @@ -524,19 +525,19 @@ class EnumParameter // def MyEnumAttr : EnumAttr; // ``` // -// By default, the assembly format of the attribute works best with operation -// assembly formats. For example: +// By default, the assembly format of the attribute wraps the symbolic value in +// angle brackets. Use the `enum` directive to print only the symbolic value in +// an operation assembly format. For example: // // ``` // def MyOp : Op { // let arguments = (ins MyEnumAttr:$enum); -// let assemblyFormat = "$enum attr-dict"; +// let assemblyFormat = "enum($enum) attr-dict"; // } // ``` // // The op will appear in the IR as `my_dialect.my_op first`. However, the -// generic format of the attribute will be `#my_dialect<"enum first">`. Override -// the attribute's assembly format as required. +// generic format of the attribute will be `#my_dialect.enum`. class EnumAttr traits = []> : AttrDef { @@ -565,9 +566,31 @@ class EnumAttr + : AttrParameter { + EnumAttr attr = enumAttr; + Dialect dialect = enumAttr.dialect; +} + +// An optional EnumAttr parameter. +class OptionalEnumAttrParameter + : EnumAttrParameter { + let defaultValue = cppStorageType # "()"; +} + +// An EnumAttr parameter with a default value. +class DefaultValuedEnumAttrParameter + : EnumAttrParameter { + let defaultValue = value; } // A property wrapping by a C++ enum. This class will automatically create bytecode diff --git a/examples/mlir/IR/OpBase.td b/examples/mlir/IR/OpBase.td index 1e34959..a130d58 100644 --- a/examples/mlir/IR/OpBase.td +++ b/examples/mlir/IR/OpBase.td @@ -98,6 +98,10 @@ def SameOperandsAndResultElementType : NativeOpTrait<"SameOperandsAndResultElementType">; // Op is a terminator. def Terminator : NativeOpTrait<"IsTerminator">; +// Op produces builtin token values. +def TokenProducerTrait : NativeOpTrait<"TokenProducerTrait">; +// Op consumes builtin token values. +def TokenConsumerTrait : NativeOpTrait<"TokenConsumerTrait">; // Op can be safely normalized in the presence of MemRefs with // non-identity maps. def MemRefsNormalizable : NativeOpTrait<"MemRefsNormalizable">; @@ -404,6 +408,11 @@ class Op props = []> { /// * void print(OpAsmPrinter &p) bit hasCustomAssemblyFormat = 0; + /// This field indicates that the operation provides a custom + /// `printProperties` hook. Setting it avoids generating the default + /// per-field `prop-dict` printer that the hook would shadow. + bit hasCustomPropertiesPrinter = 0; + // A bit indicating if the operation has additional invariants that need to // verified (aside from those verified by other ODS constructs). If set to `1`, // an additional `LogicalResult verify()` declaration will be generated on the diff --git a/examples/mlir/IR/SymbolInterfaces.td b/examples/mlir/IR/SymbolInterfaces.td index ebe0c26..97570b1 100644 --- a/examples/mlir/IR/SymbolInterfaces.td +++ b/examples/mlir/IR/SymbolInterfaces.td @@ -31,68 +31,21 @@ def Symbol : OpInterface<"SymbolOpInterface"> { let methods = [ InterfaceMethod<"Returns the name of this symbol.", - "::mlir::StringAttr", "getNameAttr", (ins), [{ - // Don't rely on the trait implementation as optional symbol operations - // may override this. - return mlir::SymbolTable::getSymbolName($_op); - }], /*defaultImplementation=*/[{ - return mlir::SymbolTable::getSymbolName(this->getOperation()); - }] + "::mlir::StringAttr", "getNameAttr", (ins) >, InterfaceMethod<"Sets the name of this symbol.", - "void", "setName", (ins "::mlir::StringAttr":$name), [{}], + "void", "setSymbolName", (ins "::mlir::StringAttr":$name), [{}], /*defaultImplementation=*/[{ - this->getOperation()->setAttr( - mlir::SymbolTable::getSymbolAttrName(), name); + auto *op = this->getOperation(); + op->setInherentAttr(::mlir::StringAttr::get( + op->getContext(), mlir::SymbolTable::getSymbolAttrName()), name); }] >, InterfaceMethod<"Gets the visibility of this symbol.", - "mlir::SymbolTable::Visibility", "getVisibility", (ins), [{}], - /*defaultImplementation=*/[{ - return mlir::SymbolTable::getSymbolVisibility(this->getOperation()); - }] - >, - InterfaceMethod<"Returns true if this symbol has nested visibility.", - "bool", "isNested", (ins), [{}], - /*defaultImplementation=*/[{ - return $_op.getVisibility() == mlir::SymbolTable::Visibility::Nested; - }] - >, - InterfaceMethod<"Returns true if this symbol has private visibility.", - "bool", "isPrivate", (ins), [{}], - /*defaultImplementation=*/[{ - return $_op.getVisibility() == mlir::SymbolTable::Visibility::Private; - }] - >, - InterfaceMethod<"Returns true if this symbol has public visibility.", - "bool", "isPublic", (ins), [{}], - /*defaultImplementation=*/[{ - return $_op.getVisibility() == mlir::SymbolTable::Visibility::Public; - }] + "mlir::SymbolTable::Visibility", "getVisibility", (ins) >, InterfaceMethod<"Sets the visibility of this symbol.", - "void", "setVisibility", (ins "mlir::SymbolTable::Visibility":$vis), [{}], - /*defaultImplementation=*/[{ - mlir::SymbolTable::setSymbolVisibility(this->getOperation(), vis); - }] - >, - InterfaceMethod<"Sets the visibility of this symbol to be nested.", - "void", "setNested", (ins), [{}], - /*defaultImplementation=*/[{ - $_op.setVisibility(mlir::SymbolTable::Visibility::Nested); - }] - >, - InterfaceMethod<"Sets the visibility of this symbol to be private.", - "void", "setPrivate", (ins), [{}], - /*defaultImplementation=*/[{ - $_op.setVisibility(mlir::SymbolTable::Visibility::Private); - }] - >, - InterfaceMethod<"Sets the visibility of this symbol to be public.", - "void", "setPublic", (ins), [{}], - /*defaultImplementation=*/[{ - $_op.setVisibility(mlir::SymbolTable::Visibility::Public); - }] + "void", "setVisibility", (ins "mlir::SymbolTable::Visibility":$vis) >, InterfaceMethod<[{ Get all of the uses of the current symbol that are nested within the @@ -163,7 +116,7 @@ def Symbol : OpInterface<"SymbolOpInterface"> { // If this is an optional symbol, bail out early if possible. auto concreteOp = cast($_op); if (concreteOp.isOptionalSymbol()) { - if(!concreteOp->getInherentAttr(::mlir::SymbolTable::getSymbolAttrName()).value_or(Attribute{})) + if (!concreteOp.getNameAttr()) return success(); } if (::mlir::failed(::mlir::detail::verifySymbol($_op))) @@ -181,23 +134,71 @@ def Symbol : OpInterface<"SymbolOpInterface"> { return success(); }]; - let extraSharedClassDeclaration = [{ - using Visibility = mlir::SymbolTable::Visibility; - + let extraClassDeclaration = [{ /// Convenience version of `getNameAttr` that returns a StringRef. ::mlir::StringRef getName() { return getNameAttr().getValue(); } - /// Convenience version of `setName` that take a StringRef. + /// Convenience version of `setSymbolName` that takes a StringRef. + void setSymbolName(::mlir::StringRef name) { + setSymbolName(::mlir::StringAttr::get(getOperation()->getContext(), name)); + } + + [[deprecated("use setSymbolName instead")]] + void setName(::mlir::StringAttr name) { + setSymbolName(name); + } + + [[deprecated("use setSymbolName instead")]] void setName(::mlir::StringRef name) { - setName(::mlir::StringAttr::get($_op->getContext(), name)); + setSymbolName(name); + } + + /// Return the conventional symbol visibility attribute name. + static ::mlir::StringRef getDefaultVisibilityAttrName() { + return "sym_visibility"; + } + + }]; + + let extraSharedClassDeclaration = [{ + using Visibility = mlir::SymbolTable::Visibility; + + /// Returns true if this symbol has nested visibility. + bool isNested() { + return $_op.getVisibility() == mlir::SymbolTable::Visibility::Nested; + } + + /// Returns true if this symbol has private visibility. + bool isPrivate() { + return $_op.getVisibility() == mlir::SymbolTable::Visibility::Private; + } + + /// Returns true if this symbol has public visibility. + bool isPublic() { + return $_op.getVisibility() == mlir::SymbolTable::Visibility::Public; + } + + /// Sets the visibility of this symbol to be nested. + void setNested() { + $_op.setVisibility(mlir::SymbolTable::Visibility::Nested); + } + + /// Sets the visibility of this symbol to be private. + void setPrivate() { + $_op.setVisibility(mlir::SymbolTable::Visibility::Private); + } + + /// Sets the visibility of this symbol to be public. + void setPublic() { + $_op.setVisibility(mlir::SymbolTable::Visibility::Public); } }]; // Add additional classof checks to properly handle "optional" symbols. let extraClassOf = [{ - return $_op->hasAttr(::mlir::SymbolTable::getSymbolAttrName()); + return static_cast($_op.getNameAttr()); }]; } @@ -245,6 +246,12 @@ def SymbolUserAttrInterface : AttrInterface<"SymbolUserAttrInterface"> { // Symbol Traits //===----------------------------------------------------------------------===// +// Op stores its symbol name in a `sym_name` inherent attribute. +def SymbolName : NativeOpTrait<"SymbolName">; + +// Op stores its symbol visibility in a `sym_visibility` inherent attribute. +def SymbolVisibility : NativeOpTrait<"SymbolVisibility">; + // Op defines a symbol table. def SymbolTable : NativeOpTrait<"SymbolTable">; diff --git a/examples/mlir/IR/TensorEncoding.td b/examples/mlir/IR/TensorEncoding.td index d8ccd1f..a9f93a9 100644 --- a/examples/mlir/IR/TensorEncoding.td +++ b/examples/mlir/IR/TensorEncoding.td @@ -23,6 +23,34 @@ def VerifiableTensorEncoding : AttrInterface<"VerifiableTensorEncoding"> { let cppNamespace = "::mlir"; let description = [{ Verifies an encoding attribute for a tensor. + + This interface also doubles as the contract some `tensor.*` + canonicalization patterns rely on when refining a tensor's shape to be + more static (e.g. folding a `tensor.cast` into a consumer, or turning a + constant `Value` operand into a static dimension). When such a pattern + needs to decide whether an existing encoding still applies to the + refined shape: + + - if the encoding implements this interface, the pattern re-runs + `verifyEncoding` against the refined shape and keeps the encoding + only if it still holds; this is how a rank- or shape-dependent + encoding (e.g. a sparse tensor encoding, which encodes a per-dimension + layout) gets dropped instead of ending up attached to a + `RankedTensorType` it is not valid for. + - if the encoding does not implement this interface, it is treated as + opaque and shape-agnostic, and is propagated onto the refined type + unconditionally. Not implementing the interface is thus an implicit + opt-out: the attribute's author is asserting that its validity does + not depend on the tensor's shape, and takes on responsibility for + that invariant. + + Patterns that instead merge or otherwise combine multiple tensors with + possibly-differing encodings (e.g. `tensor.concat`, or collapsing + dimensions via `tensor.collapse_shape`) do not attempt to verify or + propagate an encoding at all, since there is no way to statically prove + an arbitrary encoding remains valid once tensors are combined or a + dynamic dimension folds two static ones together — the encoding is + dropped from the result in that case. }]; let methods = [ InterfaceMethod< diff --git a/examples/mlir/Interfaces/CallInterfaces.td b/examples/mlir/Interfaces/CallInterfaces.td index 19d3afe..47b0f7c 100644 --- a/examples/mlir/Interfaces/CallInterfaces.td +++ b/examples/mlir/Interfaces/CallInterfaces.td @@ -77,6 +77,18 @@ def CallOpInterface : OpInterface<"CallOpInterface", another. These operations may be traditional direct calls `call @foo`, or indirect calls to other operations `call_indirect %foo`. An operation that uses this interface, must *not* also provide the `CallableOpInterface`. + + This interface distinguishes between forwarded operands/results, and + consumed/produced operands/results. Forwarded operands/results are in a + 1:1 relationship with the arguments/results of the callee: the i-th + forwarded operand is passed as the i-th argument of the callee and the i-th + forwarded result receives the i-th value returned by the callee. + Corresponding types are not required to be equal; it is up to the operation + to verify them as needed. + + Note: This interface does not model variadic operands ("argument pack"). + Neither does the CallableOpInterface. Variadic operands should be modeled + as consumed operands. }]; let cppNamespace = "::mlir"; @@ -99,17 +111,41 @@ def CallOpInterface : OpInterface<"CallOpInterface", "void", "setCalleeFromCallable", (ins "::mlir::CallInterfaceCallable":$callee) >, InterfaceMethod<[{ - Returns the operands within this call that are used as arguments to the - callee. + Returns the operands of this call that are used as arguments to the + callee ("forwarded operands"). + + The returned range must be a contiguous sub-range of the operation's + operands. The i-th forwarded operand is passed as the i-th argument of + the callee. Operands that are not part of the returned range are + consumed by the operation itself and are not passed to the callee as + arguments. }], "::mlir::Operation::operand_range", "getArgOperands" >, InterfaceMethod<[{ - Returns the operands within this call that are used as arguments to the - callee as a mutable range. + Returns the operands of this call that are used as arguments to the + callee ("forwarded operands") as a mutable range. + + This must be the same range of operands as returned by `getArgOperands`. }], "::mlir::MutableOperandRange", "getArgOperandsMutable" >, + InterfaceMethod<[{ + Returns the results of this call that receive the values returned by the + callee ("forwarded results"). + + The returned range must be a contiguous sub-range of the operation's + results. The i-th forwarded result receives the i-th value returned by + the callee. Results that are not part of the returned range are produced + by the operation itself and do not originate from the callee. + + By default, all results of the operation are forwarded results. + }], + "::mlir::Operation::result_range", "getForwardedResults", (ins), + /*methodBody=*/[{}], /*defaultImplementation=*/[{ + return $_op->getResults(); + }] + >, InterfaceMethod<[{ Resolve the callable operation for given callee to a CallableOpInterface, or nullptr if a valid callable was not resolved. diff --git a/examples/mlir/Interfaces/ControlFlowInterfaces.td b/examples/mlir/Interfaces/ControlFlowInterfaces.td index 06fa724..0808292 100644 --- a/examples/mlir/Interfaces/ControlFlowInterfaces.td +++ b/examples/mlir/Interfaces/ControlFlowInterfaces.td @@ -93,6 +93,9 @@ def BranchOpInterface : OpInterface<"BranchOpInterface"> { InterfaceMethod<[{ This method is called to compare types along control-flow edges. By default, the types are checked as equal. + + Note: Operations that do not support a certain type for successor + operands at all should return "false" if `lhs` / `rhs` is that type. }], "bool", "areTypesCompatible", (ins "::mlir::Type":$lhs, "::mlir::Type":$rhs), [{}], @@ -118,15 +121,16 @@ def BranchOpInterface : OpInterface<"BranchOpInterface"> { def RegionBranchOpInterface : OpInterface<"RegionBranchOpInterface"> { let description = [{ This interface provides information for region-holding operations that - exhibit branching behavior between held regions. I.e., this interface allows - for expressing control flow information for region holding operations. + exhibit branching behavior between held regions. It models the control flow + edges between regions (and between the op and its regions), as well as the + data flow (value propagation) that occurs along those control flow edges. This interface is meant to model well-defined cases of control-flow and value propagation, where what occurs along control-flow edges is assumed to be side-effect free. - A "region branch point" indicates a point from which a branch originates. It - can indicate: + A "region branch point" indicates the point from which a branch (edge) + originates. It can indicate: 1. A `RegionBranchTerminatorOpInterface` terminator in any of the immediately nested regions of this op. 2. `RegionBranchPoint::parent()`: the branch originates from outside of the @@ -139,7 +143,7 @@ def RegionBranchOpInterface : OpInterface<"RegionBranchOpInterface"> { A "region successor" indicates the target of a branch. It can indicate: 1. A region of this op. - 2. `RegionSuccessor::parent()`, i.e., the control flow leaves this op. + 2. A parent operation, i.e., the control flow leaves/resumes after that op. The SSA values to which successor operands are forwarded are called "successor inputs". @@ -170,8 +174,8 @@ def RegionBranchOpInterface : OpInterface<"RegionBranchOpInterface"> { `scf.for` has one region. There are two region branch points with two identical region successors: - * parent => parent(%r), region0(%a) - * `scf.yield` => parent(%r), region0(%a) + * parent => op(%r), region0(%a) + * `scf.yield` => op(%r), region0(%a) `%a` and %r are successor inputs. `%b` is an entry successor operand. `%c` is a successor operand. @@ -198,14 +202,13 @@ def RegionBranchOpInterface : OpInterface<"RegionBranchOpInterface"> { }] >, InterfaceMethod<[{ - Returns the potential region successors when first executing the op. + Returns all potential region successors when first executing the op. - Unlike `getSuccessorRegions`, this method also passes along the - constant operands of this op. Based on these, the implementation may - filter out certain successors. By default, simply dispatches to - `getSuccessorRegions`. `operands` contains an entry for every - operand of this op, with a null attribute if the operand has no constant - value. + Unlike `getSuccessorRegions`, this method also receives the constant + operands of this op (one entry per operand, "null" if the operand has + no/unknown constant value). The implementation may use this information + to filter out successors. By default, it simply dispatches to + `getSuccessorRegions`. Note: The control flow does not necessarily have to enter any region of this op. @@ -245,7 +248,7 @@ def RegionBranchOpInterface : OpInterface<"RegionBranchOpInterface"> { "::llvm::SmallVectorImpl<::mlir::RegionSuccessor> &":$regions) >, InterfaceMethod<[{ - Returns the potential region successors when branching from any + Returns all potential region successors when branching from any terminator in `region`. }], "void", "getSuccessorRegions", @@ -265,7 +268,7 @@ def RegionBranchOpInterface : OpInterface<"RegionBranchOpInterface"> { InterfaceMethod<[{ Return all successor inputs for the given region successor. If the given region successor is a region, then the returned values are block - arguments. Otherwise, if the given region successor is the "parent", + arguments. Otherwise, if the given region successor is an operation, the returned values are op results. }], "::mlir::ValueRange", "getSuccessorInputs", @@ -276,7 +279,7 @@ def RegionBranchOpInterface : OpInterface<"RegionBranchOpInterface"> { return ::mlir::ValueRange(); }]>, InterfaceMethod<[{ - Returns the potential branching points (predecessors) for a given + Returns all potential branching points (predecessors) for a given region successor. }], "void", "getPredecessors", @@ -289,15 +292,14 @@ def RegionBranchOpInterface : OpInterface<"RegionBranchOpInterface"> { ::llvm::SmallVector<::mlir::RegionSuccessor> successors; op.getSuccessorRegions(point, successors); bool isPred = llvm::any_of(successors, [&] (const auto &succ) { - return succ.getSuccessor() == successor.getSuccessor() || - (succ.isParent() && successor.isParent()); - }); + return succ == successor; + }); if (isPred) predecessors.push_back(point); } }]>, InterfaceMethod<[{ - Returns the potential values across all (predecessors) for a given successor + Returns all potential values across all (predecessors) for a given successor input, modeled by its index (its position in the list of values). }], "void", "getPredecessorValues", @@ -439,14 +441,13 @@ def RegionBranchTerminatorOpInterface : }] >, InterfaceMethod<[{ - Returns the potential region successors that are branched to after this + Returns all potential region successors that are branched to after this terminator based on the given constant operands. - This method also passes along the constant operands of this op. - `operands` contains an entry for every operand of this op, with a null - attribute if the operand has no constant value. - - The default implementation simply dispatches to the parent + This method also receives the constant operands of this op (one entry + per operand, "null" if the operand has no/unknown constant value). The + implementation may use this information to filter out successors. + By default, it simply dispatches to the parent `RegionBranchOpInterface`'s `getSuccessorRegions` implementation. }], "void", "getSuccessorRegions", diff --git a/examples/mlir/Interfaces/FunctionInterfaces.td b/examples/mlir/Interfaces/FunctionInterfaces.td index f701e82..4f312a6 100644 --- a/examples/mlir/Interfaces/FunctionInterfaces.td +++ b/examples/mlir/Interfaces/FunctionInterfaces.td @@ -132,8 +132,8 @@ def FunctionOpInterface : OpInterface<"FunctionOpInterface", [ OpBuilder &builder, OperationState &state, StringRef name, Type type, ArrayRef attrs, TypeRange inputTypes) { OpBuilder::InsertionGuard g(builder); - state.addAttribute(SymbolTable::getSymbolAttrName(), - builder.getStringAttr(name)); + state.getOrAddProperties().sym_name = + builder.getStringAttr(name); state.addAttribute(ConcreteOp::getFunctionTypeAttrName(state.name), TypeAttr::get(type)); state.attributes.append(attrs.begin(), attrs.end()); diff --git a/examples/mlir/Interfaces/InferIntDivisibilityOpInterface.td b/examples/mlir/Interfaces/InferIntDivisibilityOpInterface.td new file mode 100644 index 0000000..c665475 --- /dev/null +++ b/examples/mlir/Interfaces/InferIntDivisibilityOpInterface.td @@ -0,0 +1,41 @@ +//===- InferIntDivisibilityOpInterface.td - Integer Divisibility -*- tablegen -*-===// +// +// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. +// See https://llvm.org/LICENSE.txt for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception +// +//===----------------------------------------------------------------------===// +// +// Defines the interface for divisibility analysis on scalar integers. +// +//===----------------------------------------------------------------------===// + +#ifndef MLIR_INTERFACES_INFERINTDIVISIBILITYOPINTERFACE +#define MLIR_INTERFACES_INFERINTDIVISIBILITYOPINTERFACE + +include "mlir/IR/OpBase.td" + +def InferIntDivisibilityOpInterface : + OpInterface<"InferIntDivisibilityOpInterface"> { + let description = [{ + Allows operations to participate in integer divisibility analysis. + }]; + let cppNamespace = "::mlir"; + + let methods = [ + InterfaceMethod< + /*desc=*/[{ + Infer the divisibility of the results of this op given the + divisibility of its arguments. For each result value, the method + should call `setResultDivs` with that `Value` as an argument. + }], + /*retTy=*/"void", + /*methodName=*/"inferResultDivisibility", + /*args=*/(ins + "::llvm::ArrayRef<::mlir::IntegerDivisibility>":$argDivs, + "::mlir::SetIntDivisibilityFn":$setResultDivs) + > + ]; +} + +#endif // MLIR_INTERFACES_INFERINTDIVISIBILITYOPINTERFACE diff --git a/examples/mlir/Interfaces/LoopLikeInterface.td b/examples/mlir/Interfaces/LoopLikeInterface.td index 5fb8973..0bb82c1 100644 --- a/examples/mlir/Interfaces/LoopLikeInterface.td +++ b/examples/mlir/Interfaces/LoopLikeInterface.td @@ -62,7 +62,8 @@ def LoopLikeOpInterface : OpInterface<"LoopLikeOpInterface"> { /*args=*/(ins "::mlir::Value ":$value), /*methodBody=*/"", /*defaultImplementation=*/[{ - return !$_op->isAncestor(value.getParentRegion()->getParentOp()); + ::mlir::Region *region = value.getParentRegion(); + return !region || !region->isAttached() || !$_op->isAncestor(region->getParentOp()); }] >, InterfaceMethod<[{ diff --git a/examples/mlir/Interfaces/MemorySlotInterfaces.td b/examples/mlir/Interfaces/MemorySlotInterfaces.td index 801555f..3997c84 100644 --- a/examples/mlir/Interfaces/MemorySlotInterfaces.td +++ b/examples/mlir/Interfaces/MemorySlotInterfaces.td @@ -121,11 +121,13 @@ def PromotableMemOpInterface : OpInterface<"PromotableMemOpInterface"> { storing a value to a slot must always be able to provide the value it stores. This method is only called once per slot promotion, and only on operations that store to the slot according to the `storesTo` method. - The returned value must dominate all operations dominated by the storing - operation. - The builder is located immediately after the memory operation on call. - No IR deletion is allowed in this method. IR mutations must not + The returned value must dominate the memory operation. This ensures + that new uses inserted by `visitReplacedValues` after the memory + operation are properly dominated. + + The builder is positioned immediately before the memory operation on + call. No IR deletion is allowed in this method. IR mutations must not introduce new uses of the memory slot. Existing control flow must not be modified. }], @@ -235,29 +237,30 @@ def PromotableOpInterface : OpInterface<"PromotableOpInterface"> { "::mlir::OpBuilder &":$builder) >, InterfaceMethod<[{ - This method allows the promoted operation to visit the SSA values used - in place of the memory slot once the promotion process of the memory - slot is complete. + Indicates whether the promoted operation needs to visit the SSA values + that replace the memory slot. - If this method returns true, the `visitReplacedValues` method on this - operation will be called after the main mutation stage finishes - (i.e., after all ops have been processed with `removeBlockingUses`). + If true, `visitReplacedValues` will be called once for this operation + after all reaching definitions have been computed but before any + blocking uses are removed. - Operations should only visit the replaced values if the intended - transformation applies to all the replaced values. Furthermore, replaced - values must not be deleted. + This should only return true if the intended transformation applies + to all replaced values. Replaced values must not be deleted. }], "bool", "requiresReplacedValues", (ins), [{}], [{ return false; }] >, InterfaceMethod<[{ - Transforms the IR using the SSA values that replaced the memory slot. + Transforms the IR using the SSA values replacing the memory slot. - This method will only be called after all blocking uses have been - scheduled for removal and if `requiresReplacedValues` returned - true. + Called after all reaching definitions for the slot have been computed + but before any blocking uses are removed, provided + `requiresReplacedValues` returns true. `mutatedDefs` contains the + replacing values. Any new uses of a load result created here will be + automatically redirected to its reaching definition when the load is + replaced. - The builder is located after the promotable operation on call. During - the transformation, *no operation should be deleted*. + The builder is positioned after the promotable operation. + No operations may be deleted during this transformation. }], "void", "visitReplacedValues", (ins "::llvm::ArrayRef>":$mutatedDefs, @@ -266,6 +269,90 @@ def PromotableOpInterface : OpInterface<"PromotableOpInterface"> { ]; } +def PromotableAliaserInterface : OpInterface<"PromotableAliaserInterface"> { + let description = [{ + Describes an operation that creates a transparent alias of a memory slot + accessed through one of its operands. Mem2Reg traverses chains of these + aliases to project slot values across them. This allows load and store + operations on the alias to be promoted as if they were directly accessing + the underlying slot. + + Since an alias remains a blocking use of the underlying slot pointer, the + operation must also implement either `PromotableOpInterface` or + `PromotableMemOpInterface`. This ensures that mem2reg can remove the alias + after the slot has been promoted. + }]; + let cppNamespace = "::mlir"; + + let verify = [{ + if (!::mlir::isa<::mlir::PromotableOpInterface, + ::mlir::PromotableMemOpInterface>($_op)) + return $_op->emitOpError( + "implements `PromotableAliaserInterface` but must also implement " + "`PromotableOpInterface` or `PromotableMemOpInterface`."); + return ::mlir::success(); + }]; + + let methods = [ + InterfaceMethod<[{ + Populates `newMemorySlots` with the memory slots this operation + exposes by aliasing `parentSlot` (accessed via + `aliasedSlotPointerOperand`). Each new slot's pointer must be a + result of this operation, and its element type may differ from the + parent's. Leave the vector empty if no alias is exposed. An operation + can expose multiple aliases for the same parent slot. + + `parentSlot` is provided so that aliasers using opaque pointers can + derive the new slot's element type from `parentSlot.elemType`. + + Exposing an alias requires implementing the two projection methods + below to bridge values between the parent and new slot element types. + If these projections cannot be performed, `newMemorySlots` must be left + empty. + + No IR mutation is allowed in this method. + }], + "void", + "getPromotableSlotAliases", + (ins "::mlir::OpOperand &":$aliasedSlotPointerOperand, + "const ::mlir::MemorySlot &":$parentSlot, + "::llvm::SmallVectorImpl<::mlir::MemorySlot> &":$newMemorySlots) + >, + InterfaceMethod<[{ + Extracts the value of `aliasSlot` from `slotValue` (the value of + `parentSlot`). Mem2Reg invokes this method when a load on the new slot + requires the parent's value to be materialized with the new slot's + element type. + }], + "::mlir::Value", + "projectSlotValueToAliasValue", + (ins "::mlir::OpOperand &":$aliasedSlotPointerOperand, + "const ::mlir::MemorySlot &":$parentSlot, + "const ::mlir::MemorySlot &":$aliasSlot, + "::mlir::Value":$slotValue, + "::mlir::OpBuilder &":$builder), [{}], + [{ return slotValue; }] + >, + InterfaceMethod<[{ + Reconstructs the value of `parentSlot` from `aliasValue` (a store to + `aliasSlot`) and `reachingDef` (the parent slot's value immediately + preceding the store). For full aliases, `reachingDef` can be ignored. + For partial sub-aliases, it allows the result to be constructed by + inserting `aliasValue` into `reachingDef`. + }], + "::mlir::Value", + "projectAliasValueToSlotValue", + (ins "::mlir::OpOperand &":$aliasedSlotPointerOperand, + "const ::mlir::MemorySlot &":$parentSlot, + "const ::mlir::MemorySlot &":$aliasSlot, + "::mlir::Value":$aliasValue, + "::mlir::Value":$reachingDef, + "::mlir::OpBuilder &":$builder), [{}], + [{ return aliasValue; }] + >, + ]; +} + def PromotableRegionOpInterface : OpInterface<"PromotableRegionOpInterface"> { let description = [{ diff --git a/examples/mlir/Interfaces/SideEffectInterfaceBase.td b/examples/mlir/Interfaces/SideEffectInterfaceBase.td index 043829c..c1b7372 100644 --- a/examples/mlir/Interfaces/SideEffectInterfaceBase.td +++ b/examples/mlir/Interfaces/SideEffectInterfaceBase.td @@ -22,9 +22,12 @@ include "mlir/IR/OpBase.td" //===----------------------------------------------------------------------===// // A generic resource that can be attached to a general base side effect. -class Resource { +class Resource { /// The resource that the associated effect is being applied to. string name = resourceName; + + /// The optional effect parameters attribute. + code parameters = resourceParameters; } // An intrinsic resource that lives in the ::mlir::SideEffects namespace. @@ -177,6 +180,9 @@ class SideEffect { description below): - `getTiledImplementationFromOperandTiles` - `getIterationDomainTileFromOperandTiles`. + + `getTiledImplementation`, `generateResultTileValue`, + `getTiledImplementationFromOperandTiles` and + `getIterationDomainTileFromOperandTiles` each have a second, overloaded form + taking an extra `ArrayRef` caller hint (see + `InnerTileAlignment` in TilingInterface.h). The hint-bearing overload + defaults to forwarding to the hint-less one, so only ops that opt in consult + it. Because the two forms share a source name, an implementer that defines + only the hint-less overload hides the inherited defaulted one (C++ name + hiding) and fails to compile when the interface forwards to it. }]; let cppNamespace = "::mlir"; let methods = [ @@ -115,6 +125,28 @@ def TilingInterface : OpInterface<"TilingInterface"> { return {}; }] >, + InterfaceMethod< + /*desc=*/[{ + Variant of `getTiledImplementation` that additionally takes a + per-iteration-domain `InnerTileAlignment` hint (see + `InnerTileAlignment`) asserting how each loop tile size relates to a + pack/unpack inner tile. The hint is consulted only by pack/unpack + implementations and may be empty (all `Unknown`). The default + implementation ignores it and forwards to the hint-less overload. + }], + /*retType=*/"::mlir::FailureOr<::mlir::TilingResult>", + /*methodName=*/"getTiledImplementation", + /*args=*/(ins + "::mlir::OpBuilder &":$b, + "::mlir::ArrayRef<::mlir::OpFoldResult> ":$offsets, + "::mlir::ArrayRef<::mlir::OpFoldResult> ":$sizes, + "::mlir::ArrayRef<::mlir::InnerTileAlignment> ":$innerTileAlignments), + /*methodBody=*/"", + /*defaultImplementation=*/[{ + return ::mlir::cast<::mlir::TilingInterface>($_op.getOperation()) + .getTiledImplementation(b, offsets, sizes); + }] + >, InterfaceMethod< /*desc=*/[{ Method to return the position of the result tile computed by the @@ -199,6 +231,28 @@ def TilingInterface : OpInterface<"TilingInterface"> { return failure(); }] >, + InterfaceMethod< + /*desc=*/[{ + Variant of `generateResultTileValue` that additionally takes a + per-iteration-domain `InnerTileAlignment` hint (see + `InnerTileAlignment`). The hint is consulted only by pack/unpack + implementations and may be empty (all `Unknown`). The default + implementation ignores it and forwards to the hint-less overload. + }], + /*retType=*/"::mlir::FailureOr<::mlir::TilingResult>", + /*methodName=*/"generateResultTileValue", + /*args=*/(ins + "::mlir::OpBuilder &":$b, + "unsigned":$resultNumber, + "::mlir::ArrayRef<::mlir::OpFoldResult>":$offsets, + "::mlir::ArrayRef<::mlir::OpFoldResult>":$sizes, + "::mlir::ArrayRef<::mlir::InnerTileAlignment>":$innerTileAlignments), + /*methodBody=*/"", + /*defaultImplementation=*/[{ + return ::mlir::cast<::mlir::TilingInterface>($_op.getOperation()) + .generateResultTileValue(b, resultNumber, offsets, sizes); + }] + >, InterfaceMethod< /*desc=*/[{ Method to generate the tiled implementation of an operation that uses @@ -229,6 +283,29 @@ def TilingInterface : OpInterface<"TilingInterface"> { return failure(); }] >, + InterfaceMethod< + /*desc=*/[{ + Variant of `getTiledImplementationFromOperandTiles` that additionally + takes a per-iteration-domain `InnerTileAlignment` hint (see + `InnerTileAlignment`). The hint is consulted only by pack/unpack + implementations and may be empty (all `Unknown`). The default + implementation ignores it and forwards to the hint-less overload. + }], + /*retType=*/"::mlir::FailureOr<::mlir::TilingResult>", + /*methodName=*/"getTiledImplementationFromOperandTiles", + /*args=*/(ins + "::mlir::OpBuilder &":$b, + "::mlir::ArrayRef":$operandNumbers, + "::mlir::ArrayRef<::mlir::SmallVector<::mlir::OpFoldResult>>":$allOffsets, + "::mlir::ArrayRef<::mlir::SmallVector<::mlir::OpFoldResult>>":$allSizes, + "::mlir::ArrayRef<::mlir::InnerTileAlignment>":$innerTileAlignments), + /*methodBody=*/"", + /*defaultImplementation=*/[{ + return ::mlir::cast<::mlir::TilingInterface>($_op.getOperation()) + .getTiledImplementationFromOperandTiles(b, operandNumbers, + allOffsets, allSizes); + }] + >, InterfaceMethod< /*desc=*/[{ Method to return the tile of the iteration domain that uses a given @@ -304,6 +381,33 @@ def TilingInterface : OpInterface<"TilingInterface"> { return failure(); }] >, + InterfaceMethod< + /*desc=*/[{ + Variant of `getIterationDomainTileFromOperandTiles` that additionally + takes a per-iteration-domain `InnerTileAlignment` hint (see + `InnerTileAlignment`). The hint is consulted only by pack/unpack + implementations and may be empty (all `Unknown`). The default + implementation ignores it and forwards to the hint-less overload. + }], + /*retType=*/"::llvm::LogicalResult", + /*methodName=*/"getIterationDomainTileFromOperandTiles", + /*args=*/(ins + "::mlir::OpBuilder &":$b, + "::mlir::ArrayRef":$operandNumbers, + "::mlir::ArrayRef<::mlir::SmallVector<::mlir::OpFoldResult>> ":$allOffsets, + "::mlir::ArrayRef<::mlir::SmallVector<::mlir::OpFoldResult>> ":$allSizes, + "::mlir::SmallVectorImpl<::mlir::OpFoldResult> &":$iterDomainOffsets, + "::mlir::SmallVectorImpl<::mlir::OpFoldResult> &":$iterDomainSizes, + "::mlir::ArrayRef<::mlir::InnerTileAlignment> ":$innerTileAlignments), + /*methodBody=*/"", + /*defaultImplementation=*/[{ + return ::mlir::cast<::mlir::TilingInterface>($_op.getOperation()) + .getIterationDomainTileFromOperandTiles(b, operandNumbers, + allOffsets, allSizes, + iterDomainOffsets, + iterDomainSizes); + }] + >, InterfaceMethod< /*desc=*/[{ Method to return the tile of the iteration domain based diff --git a/examples/mlir/_tblgen-regress/attr-or-type-format.td b/examples/mlir/_tblgen-regress/attr-or-type-format.td index 3a46459..d908a2a 100644 --- a/examples/mlir/_tblgen-regress/attr-or-type-format.td +++ b/examples/mlir/_tblgen-regress/attr-or-type-format.td @@ -229,6 +229,62 @@ def EnumAttrB : EnumAttr { let assemblyFormat = "$value"; } +def TestStructEnum : I32EnumAttr<"TestStructEnum", "TestStructEnumType", [ + I32EnumAttrCase<"first", 0>, + I32EnumAttrCase<"second", 1>, +]> { + let genSpecializedAttr = 0; +} + +def EnumAttrC : EnumAttr; + +// EnumAttr parameters in a struct use the underlying enum syntax rather than +// the EnumAttr's assembly format. +// ATTR-LABEL: AttrWithEnumAttrStructAttr::parse +// ATTR: _result_required = [&]() -> ::mlir::FailureOr<::TestStructEnumAttr> { +// ATTR-NEXT: auto odsEnumValue = ::mlir::FieldParser<::TestStructEnum>::parse(odsParser); +// ATTR: return ::TestStructEnumAttr::get(odsParser.getContext(), *odsEnumValue); +// ATTR-LABEL: AttrWithEnumAttrStructAttr::print +// ATTR: odsPrinter << getRequired().getValue(); +// ATTR: odsPrinter << getOptional().getValue(); +def AttrWithEnumAttrStruct : TestAttr<"AttrWithEnumAttrStruct"> { + let parameters = (ins + EnumAttrParameter:$required, + OptionalEnumAttrParameter:$optional + ); + let mnemonic = "enum_attr_struct"; + let assemblyFormat = "`<` struct(params) `>`"; +} + +def TestStructBitEnum : I32BitEnum<"TestStructBitEnum", "", [ + I32BitEnumCaseNone<"none">, + I32BitEnumCaseBit<"read", 0>, + I32BitEnumCaseBit<"write", 1>, +]> { + let separator = ", "; +} + +def EnumAttrD : EnumAttr; + +// A non-final comma-separated bit enum is bracketed so its commas are not +// confused with the struct separator. +// ATTR-LABEL: AttrWithBitEnumAttrStructAttr::parse +// ATTR: if (odsParser.parseLSquare()) return {}; +// ATTR: FieldParser<::TestStructBitEnum>::parse(odsParser) +// ATTR: if (odsParser.parseRSquare()) return {}; +// ATTR-LABEL: AttrWithBitEnumAttrStructAttr::print +// ATTR: odsPrinter << "["; +// ATTR: odsPrinter << getFlags().getValue(); +// ATTR: odsPrinter << "]"; +def AttrWithBitEnumAttrStruct : TestAttr<"AttrWithBitEnumAttrStruct"> { + let parameters = (ins + EnumAttrParameter:$flags, + "int64_t":$count + ); + let mnemonic = "bit_enum_attr_struct"; + let assemblyFormat = "`<` struct(params) `>`"; +} + /// Test type parser and printer that mix variables and struct are generated /// correctly. diff --git a/examples/mlir/_tblgen-regress/constraint-unique.td b/examples/mlir/_tblgen-regress/constraint-unique.td index 3f2e5cd..55d6a56 100644 --- a/examples/mlir/_tblgen-regress/constraint-unique.td +++ b/examples/mlir/_tblgen-regress/constraint-unique.td @@ -17,7 +17,7 @@ def OtherType : Type; def AnAttrPred : CPred<"attrPred($_self, $_op)">; def AnAttr : Attr; -def OtherAttr : Attr; +def OtherAttr : Attr; def ASuccessorPred : CPred<"successorPred($_self, $_op)">; def ASuccessor : Successor; @@ -81,22 +81,22 @@ def OpC : NS_Op<"op_c"> { // CHECK: static ::llvm::LogicalResult [[$O_ATTR_CONSTRAINT:__mlir_ods_local_attr_constraint.*]]( // CHECK: if (attr && !((attrPred(attr, *op)))) // CHECK-NEXT: return emitError() << "attribute '" << attrName -// CHECK-NEXT: << "' failed to satisfy constraint: another attribute"; +// CHECK-NEXT: << "' failed to satisfy constraint: another attribute " << reformat(attr) << ""; /// Test that a successor contraint was generated. // CHECK: static ::llvm::LogicalResult [[$A_SUCCESSOR_CONSTRAINT:__mlir_ods_local_successor_constraint.*]]( // CHECK: if (!((successorPred(successor, *op)))) { // CHECK-NEXT: return op->emitOpError("successor #") << successorIndex << " ('" -// CHECK-NEXT: << successorName << ")' failed to verify constraint: a successor"; +// CHECK-NEXT: << successorName << "') failed to verify constraint: a successor"; /// Test that duplicate successor constraint was not generated. -// CHECK-NOT: << successorName << ")' failed to verify constraint: a successor"; +// CHECK-NOT: << successorName << "') failed to verify constraint: a successor"; /// Test that a successor constraint with a different description was generated. // CHECK: static ::llvm::LogicalResult [[$O_SUCCESSOR_CONSTRAINT:__mlir_ods_local_successor_constraint.*]]( // CHECK: if (!((successorPred(successor, *op)))) { // CHECK-NEXT: return op->emitOpError("successor #") << successorIndex << " ('" -// CHECK-NEXT: << successorName << ")' failed to verify constraint: another successor"; +// CHECK-NEXT: << successorName << "') failed to verify constraint: another successor"; /// Test that a region contraint was generated. // CHECK: static ::llvm::LogicalResult [[$A_REGION_CONSTRAINT:__mlir_ods_local_region_constraint.*]]( @@ -131,8 +131,7 @@ def OpC : NS_Op<"op_c"> { // CHECK: for (auto ®ion : ::llvm::MutableArrayRef((*this)->getRegion(0))) // CHECK-NEXT: if (::mlir::failed([[$A_REGION_CONSTRAINT]](*this, region, "d", index++))) // CHECK-NEXT: return ::mlir::failure(); -// CHECK: for (auto *successor : ::llvm::MutableArrayRef(c())) -// CHECK-NEXT: if (::mlir::failed([[$A_SUCCESSOR_CONSTRAINT]](*this, successor, "c", index++))) +// CHECK: if (::mlir::failed([[$A_SUCCESSOR_CONSTRAINT]](*this, getC(), "c", index++))) // CHECK-NEXT: return ::mlir::failure(); /// Test that the op with the same predicates but different with descriptions @@ -152,6 +151,5 @@ def OpC : NS_Op<"op_c"> { // CHECK: for (auto ®ion : ::llvm::MutableArrayRef((*this)->getRegion(0))) // CHECK-NEXT: if (::mlir::failed([[$O_REGION_CONSTRAINT]](*this, region, "d", index++))) // CHECK-NEXT: return ::mlir::failure(); -// CHECK: for (auto *successor : ::llvm::MutableArrayRef(c())) -// CHECK-NEXT: if (::mlir::failed([[$O_SUCCESSOR_CONSTRAINT]](*this, successor, "c", index++))) +// CHECK: if (::mlir::failed([[$O_SUCCESSOR_CONSTRAINT]](*this, getC(), "c", index++))) // CHECK-NEXT: return ::mlir::failure(); diff --git a/examples/mlir/_tblgen-regress/enums-gen.td b/examples/mlir/_tblgen-regress/enums-gen.td index cf66ad4..e64e50b 100644 --- a/examples/mlir/_tblgen-regress/enums-gen.td +++ b/examples/mlir/_tblgen-regress/enums-gen.td @@ -45,6 +45,9 @@ def MyBitEnum: I32BitEnumAttr<"MyBitEnum", "An example bit enum", // DECL: return parser.emitError(loc, "expected one of [none, tagged, Bit1, Bit2, Bit3, BitGroup] for An example bit enum, got: ") << enumKeyword; // DECL: } +// DECL: struct FieldParser, std::optional<::MyBitEnum>> { +// DECL: static constexpr bool isKeyValueCompositional = false; + // DECL: inline ::llvm::raw_ostream &operator<<(::llvm::raw_ostream &p, ::MyBitEnum value) { // DECL: auto valueStr = stringifyEnum(value); // DECL: switch (value) { @@ -58,6 +61,11 @@ def MyBitEnum: I32BitEnumAttr<"MyBitEnum", "An example bit enum", // DECL: return p << '"' << valueStr << '"'; // DECL: return p << valueStr; +// DECL: struct FieldParser<::MyCommaSeparatedBitEnum, ::MyCommaSeparatedBitEnum> { +// DECL: static constexpr bool isKeyValueCompositional = false; +// DECL: struct FieldParser, std::optional<::MyCommaSeparatedBitEnum>> { +// DECL: static constexpr bool isKeyValueCompositional = false; + // DECL: enum class MyI8Enum : uint8_t { // DECL: a = 254, // DECL: b = 255, @@ -129,6 +137,7 @@ def MyNonQuotedPrintBitEnum [None, Bit0, Bit1, Bit2, Bit3, BitGroup]>; // DECL: struct FieldParser<::MyNonQuotedPrintBitEnum, ::MyNonQuotedPrintBitEnum> { +// DECL: static constexpr bool isKeyValueCompositional = true; // DECL: template // DECL: static FailureOr<::MyNonQuotedPrintBitEnum> parse(ParserT &parser) { // DECL: ::MyNonQuotedPrintBitEnum flags = {}; @@ -149,6 +158,12 @@ def MyNonQuotedPrintBitEnum // DECL: return flags; // DECL: } +def MyCommaSeparatedBitEnum + : I32BitEnum<"MyCommaSeparatedBitEnum", "Comma-separated bit enum", + [None, Bit0, Bit1]> { + let separator = ", "; +} + // DECL: inline ::llvm::raw_ostream &operator<<(::llvm::raw_ostream &p, ::MyNonQuotedPrintBitEnum value) { // DECL: auto valueStr = stringifyEnum(value); // DECL-NEXT: return p << valueStr; diff --git a/examples/mlir/_tblgen-regress/enums-python-bindings.td b/examples/mlir/_tblgen-regress/enums-python-bindings.td index 74b9f51..487b60d 100644 --- a/examples/mlir/_tblgen-regress/enums-python-bindings.td +++ b/examples/mlir/_tblgen-regress/enums-python-bindings.td @@ -108,8 +108,8 @@ def TestBitEnum_Attr : EnumAttr; // CHECK: @register_attribute_builder("TestDialect.TestBitEnum_Attr") // CHECK: def _testbitenum_attr(x, context): -// CHECK: return _ods_ir.Attribute.parse(f'#TestDialect', context=context) +// CHECK: return _ods_ir.Attribute.parse(f'#TestDialect.testbitenum<{str(x)}>', context=context) // CHECK: @register_attribute_builder("TestDialect.TestMyEnum_Attr") // CHECK: def _testmyenum_attr(x, context): -// CHECK: return _ods_ir.Attribute.parse(f'#TestDialect', context=context) +// CHECK: return _ods_ir.Attribute.parse(f'#TestDialect.enum<{str(x)}>', context=context) diff --git a/examples/mlir/_tblgen-regress/gen-dialect-doc.td b/examples/mlir/_tblgen-regress/gen-dialect-doc.td index 7291670..b279b51 100644 --- a/examples/mlir/_tblgen-regress/gen-dialect-doc.td +++ b/examples/mlir/_tblgen-regress/gen-dialect-doc.td @@ -67,6 +67,44 @@ def TestAttrDefParams : AttrDef { let assemblyFormat = "`<` $value `>`"; } +def TestEnumForParam : + I32EnumAttr<"TestEnumForParam", + "enum for param test", [ + I32EnumAttrCase<"Alpha", 0, "alpha">, + I32EnumAttrCase<"Beta", 1, "beta">]> { + let genSpecializedAttr = 0; + let cppNamespace = "NS"; +} + +def TestAttrWithEnum : AttrDef { + let mnemonic = "with_enum"; + let parameters = (ins + "int":$value, + EnumParameter:$mode + ); + let assemblyFormat = "`<` $value `,` $mode `>`"; +} + +def TestEnumWithAttrWrapper : + I32EnumAttr<"TestEnumWithWrapper", + "enum with attr wrapper", [ + I32EnumAttrCase<"Red", 0, "red">, + I32EnumAttrCase<"Green", 1, "green">, + I32EnumAttrCase<"Blue", 2, "blue">]> { + let genSpecializedAttr = 0; + let cppNamespace = "NS"; +} +def TestEnumWithWrapperAttr : EnumAttr; + +def TestAttrWithWrappedEnum : AttrDef { + let mnemonic = "with_wrapped_enum"; + let parameters = (ins + "int":$id, + EnumParameter:$color + ); + let assemblyFormat = "`<` $id `,` $color `>`"; +} + def TestTypeDef : TypeDef { let mnemonic = "test_type_def"; } @@ -140,6 +178,20 @@ def TestEnum : // CHECK: Syntax: // CHECK: #test.test_attr_def_params +// CHECK: TestAttrWithEnumAttr +// CHECK: Syntax: +// CHECK: #test.with_enum< +// CHECK-NEXT: int, # value +// CHECK-NEXT: `alpha` | `beta` # mode +// CHECK-NEXT: > + +// CHECK: TestAttrWithWrappedEnumAttr +// CHECK: Syntax: +// CHECK: #test.with_wrapped_enum< +// CHECK-NEXT: int, # id +// CHECK-NEXT: `red` | `green` | `blue` # color +// CHECK-NEXT: > + // CHECK: ## Type constraints // CHECK: ### type summary // CHECK: type description diff --git a/examples/mlir/_tblgen-regress/op-attribute.td b/examples/mlir/_tblgen-regress/op-attribute.td index cfdaaeb..ebba86d 100644 --- a/examples/mlir/_tblgen-regress/op-attribute.td +++ b/examples/mlir/_tblgen-regress/op-attribute.td @@ -163,7 +163,7 @@ def AOp : NS_Op<"a_op", []> { // DEF: void AOp::build( // DEF: ::llvm::ArrayRef<::mlir::NamedAttribute> attributes -// DEF: odsState.addAttributes(attributes); +// DEF: buildPropertiesAndDiscardableAttributes(odsState, attributes); // DEF: void AOp::build( // DEF-SAME: const Properties &properties, @@ -283,7 +283,7 @@ def AgetOp : Op { // DEF: void AgetOp::build( // DEF: ::llvm::ArrayRef<::mlir::NamedAttribute> attributes -// DEF: odsState.addAttributes(attributes); +// DEF: buildPropertiesAndDiscardableAttributes(odsState, attributes); // DEF: void AgetOp::build( // DEF-SAME: const Properties &properties diff --git a/examples/mlir/_tblgen-regress/op-decl-and-defs.td b/examples/mlir/_tblgen-regress/op-decl-and-defs.td index e92b404..32c4bb8 100644 --- a/examples/mlir/_tblgen-regress/op-decl-and-defs.td +++ b/examples/mlir/_tblgen-regress/op-decl-and-defs.td @@ -135,23 +135,40 @@ def NS_AOp : NS_Op<"a_op", [IsolatedFromAbove, IsolatedFromAbove]> { // CHECK: static void build(::mlir::OpBuilder &odsBuilder, ::mlir::OperationState &odsState, ::mlir::TypeRange resultTypes, ::mlir::Value a, ::mlir::ValueRange b, uint32_t attr1, /*optional*/::mlir::FloatAttr some_attr2, unsigned someRegionsCount); // CHECK: static AOp create(::mlir::OpBuilder &builder, ::mlir::Location location, ::mlir::TypeRange resultTypes, ::mlir::Value a, ::mlir::ValueRange b, uint32_t attr1, /*optional*/::mlir::FloatAttr some_attr2, unsigned someRegionsCount); // CHECK: static AOp create(::mlir::ImplicitLocOpBuilder &builder, ::mlir::TypeRange resultTypes, ::mlir::Value a, ::mlir::ValueRange b, uint32_t attr1, /*optional*/::mlir::FloatAttr some_attr2, unsigned someRegionsCount); -// CHECK: static void build(::mlir::OpBuilder &, ::mlir::OperationState &odsState, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, ::llvm::ArrayRef<::mlir::NamedAttribute> attributes, unsigned numRegions) -// CHECK: static AOp create(::mlir::OpBuilder &builder, ::mlir::Location location, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, ::llvm::ArrayRef<::mlir::NamedAttribute> attributes, unsigned numRegions) -// CHECK: static AOp create(::mlir::ImplicitLocOpBuilder &builder, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, ::llvm::ArrayRef<::mlir::NamedAttribute> attributes, unsigned numRegions) -// CHECK: static void build(::mlir::OpBuilder &, ::mlir::OperationState &odsState, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, const Properties &properties, ::llvm::ArrayRef<::mlir::NamedAttribute> discardableAttributes, unsigned numRegions) -// CHECK: static AOp create(::mlir::OpBuilder &builder, ::mlir::Location location, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, const Properties &properties, ::llvm::ArrayRef<::mlir::NamedAttribute> discardableAttributes, unsigned numRegions) -// CHECK: static AOp create(::mlir::ImplicitLocOpBuilder &builder, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, const Properties &properties, ::llvm::ArrayRef<::mlir::NamedAttribute> discardableAttributes, unsigned numRegions) +// CHECK{LITERAL}: [[deprecated("use the overload taking Properties and discardableAttributes instead")]] +// CHECK-NEXT: static void build(::mlir::OpBuilder &, ::mlir::OperationState &odsState, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, ::llvm::ArrayRef<::mlir::NamedAttribute> attributes, unsigned numRegions) +// CHECK{LITERAL}: [[deprecated("use the overload taking Properties and discardableAttributes instead")]] +// CHECK-NEXT: static AOp create(::mlir::OpBuilder &builder, ::mlir::Location location, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, ::llvm::ArrayRef<::mlir::NamedAttribute> attributes, unsigned numRegions) +// CHECK{LITERAL}: [[deprecated("use the overload taking Properties and discardableAttributes instead")]] +// CHECK-NEXT: static AOp create(::mlir::ImplicitLocOpBuilder &builder, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, ::llvm::ArrayRef<::mlir::NamedAttribute> attributes, unsigned numRegions) +// CHECK-NEXT: static void build(::mlir::OpBuilder &, ::mlir::OperationState &odsState, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, const Properties &properties, ::llvm::ArrayRef<::mlir::NamedAttribute> discardableAttributes, unsigned numRegions) +// CHECK-NEXT: static AOp create(::mlir::OpBuilder &builder, ::mlir::Location location, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, const Properties &properties, ::llvm::ArrayRef<::mlir::NamedAttribute> discardableAttributes, unsigned numRegions) +// CHECK-NEXT: static AOp create(::mlir::ImplicitLocOpBuilder &builder, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, const Properties &properties, ::llvm::ArrayRef<::mlir::NamedAttribute> discardableAttributes, unsigned numRegions) // CHECK: static ::mlir::ParseResult parse(::mlir::OpAsmParser &parser, ::mlir::OperationState &result); // CHECK: void print(::mlir::OpAsmPrinter &p); // CHECK: ::llvm::LogicalResult verifyInvariants(); // CHECK: static void getCanonicalizationPatterns(::mlir::RewritePatternSet &results, ::mlir::MLIRContext *context); // CHECK: ::llvm::LogicalResult fold(FoldAdaptor adaptor, ::llvm::SmallVectorImpl<::mlir::OpFoldResult> &results); // CHECK: static ::llvm::LogicalResult setPropertiesFromParsedAttr(Properties &prop, ::mlir::Attribute attr, ::llvm::function_ref<::mlir::InFlightDiagnostic()> emitError); +// CHECK: private: +// CHECK: static void buildPropertiesAndDiscardableAttributes(::mlir::OperationState &odsState, ::llvm::ArrayRef<::mlir::NamedAttribute> attributes); +// CHECK-NEXT: public: // CHECK: // Display a graph for debugging purposes. // CHECK: void displayGraph(); // CHECK: }; // DEFS-LABEL: NS::AOp definitions +// DEFS: void AOp::build(::mlir::OpBuilder &, ::mlir::OperationState &odsState, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, ::llvm::ArrayRef<::mlir::NamedAttribute> attributes, unsigned numRegions) { +// DEFS: buildPropertiesAndDiscardableAttributes(odsState, attributes); +// DEFS: void AOp::build(::mlir::OpBuilder &, ::mlir::OperationState &odsState, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, const Properties &properties, ::llvm::ArrayRef<::mlir::NamedAttribute> discardableAttributes, unsigned numRegions) { +// DEFS: odsState.useProperties(const_cast(properties)); +// DEFS-LABEL: void AOp::buildPropertiesAndDiscardableAttributes +// DEFS: Properties &properties = odsState.getOrAddProperties(); +// DEFS-NEXT: populateDefaultProperties(odsState.name, properties); +// DEFS: if (name == "attr1" || name == "some_attr2") +// DEFS: inherentAttributes.push_back(attr); +// DEFS: odsState.addAttribute(attr.getName(), attr.getValue()); +// DEFS: if (::mlir::failed(setPropertiesFromAttr( // Check that `getAttrDictionary()` is used when not using properties. @@ -236,15 +253,41 @@ def NS_FOp : NS_Op<"op_with_all_types_constraint", // DEFS: FOp FOp::create(::mlir::OpBuilder &builder, ::mlir::Location location, ::mlir::Value a) { // DEFS: ::mlir::OperationState __state__(location, getOperationName()); // DEFS: build(builder, __state__, std::forward(a)); -// DEFS: auto __res__ = ::llvm::dyn_cast(builder.create(__state__)); -// DEFS: assert(__res__ && "builder didn't return the right type"); -// DEFS: return __res__; +// DEFS: auto __res__ = builder.create(__state__); +// DEFS: assert((::llvm::isa(__res__)) && "builder didn't return the right type"); +// DEFS: return ::llvm::cast(__res__); // DEFS: } // DEFS: FOp FOp::create(::mlir::ImplicitLocOpBuilder &builder, ::mlir::Value a) { // DEFS: return create(builder, builder.getLoc(), std::forward(a)); // DEFS: } +def NS_FirstAttrDerivedOp : NS_Op<"first_attr_derived", + [FirstAttrDerivedResultType]> { + let arguments = (ins TypeAttr:$type, AnyType:$input); + let results = (outs AnyType:$result); +} + +// CHECK-LABEL: class FirstAttrDerivedOp : +// CHECK{LITERAL}: [[deprecated("use the overload taking Properties and discardableAttributes instead")]] +// CHECK-NEXT: static void build(::mlir::OpBuilder &, ::mlir::OperationState &odsState, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, ::llvm::ArrayRef<::mlir::NamedAttribute> attributes = {}); +// CHECK{LITERAL}: [[deprecated("use the overload taking Properties and discardableAttributes instead")]] +// CHECK-NEXT: static FirstAttrDerivedOp create(::mlir::OpBuilder &builder, ::mlir::Location location, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, ::llvm::ArrayRef<::mlir::NamedAttribute> attributes = {}); +// CHECK{LITERAL}: [[deprecated("use the overload taking Properties and discardableAttributes instead")]] +// CHECK-NEXT: static FirstAttrDerivedOp create(::mlir::ImplicitLocOpBuilder &builder, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, ::llvm::ArrayRef<::mlir::NamedAttribute> attributes = {}); +// CHECK-NEXT: static void build(::mlir::OpBuilder &, ::mlir::OperationState &odsState, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, const Properties &properties, ::llvm::ArrayRef<::mlir::NamedAttribute> discardableAttributes = {}); +// CHECK-NEXT: static FirstAttrDerivedOp create(::mlir::OpBuilder &builder, ::mlir::Location location, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, const Properties &properties, ::llvm::ArrayRef<::mlir::NamedAttribute> discardableAttributes = {}); +// CHECK-NEXT: static FirstAttrDerivedOp create(::mlir::ImplicitLocOpBuilder &builder, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, const Properties &properties, ::llvm::ArrayRef<::mlir::NamedAttribute> discardableAttributes = {}); +// CHECK{LITERAL}: [[deprecated("use the overload taking Properties and discardableAttributes instead")]] +// CHECK-NEXT: static void build(::mlir::OpBuilder &odsBuilder, ::mlir::OperationState &odsState, ::mlir::ValueRange operands, ::llvm::ArrayRef<::mlir::NamedAttribute> attributes = {}); +// CHECK{LITERAL}: [[deprecated("use the overload taking Properties and discardableAttributes instead")]] +// CHECK-NEXT: static FirstAttrDerivedOp create(::mlir::OpBuilder &builder, ::mlir::Location location, ::mlir::ValueRange operands, ::llvm::ArrayRef<::mlir::NamedAttribute> attributes = {}); +// CHECK{LITERAL}: [[deprecated("use the overload taking Properties and discardableAttributes instead")]] +// CHECK-NEXT: static FirstAttrDerivedOp create(::mlir::ImplicitLocOpBuilder &builder, ::mlir::ValueRange operands, ::llvm::ArrayRef<::mlir::NamedAttribute> attributes = {}); +// CHECK-NEXT: static void build(::mlir::OpBuilder &odsBuilder, ::mlir::OperationState &odsState, ::mlir::ValueRange operands, const Properties &properties, ::llvm::ArrayRef<::mlir::NamedAttribute> discardableAttributes = {}); +// CHECK-NEXT: static FirstAttrDerivedOp create(::mlir::OpBuilder &builder, ::mlir::Location location, ::mlir::ValueRange operands, const Properties &properties, ::llvm::ArrayRef<::mlir::NamedAttribute> discardableAttributes = {}); +// CHECK-NEXT: static FirstAttrDerivedOp create(::mlir::ImplicitLocOpBuilder &builder, ::mlir::ValueRange operands, const Properties &properties, ::llvm::ArrayRef<::mlir::NamedAttribute> discardableAttributes = {}); + def NS_GOp : NS_Op<"op_with_fixed_return_type", []> { let arguments = (ins AnyType:$a); let results = (outs I32:$b); @@ -286,6 +329,7 @@ def NS_HCollectiveParamsSuppress0Op : NS_Op<"op_collective_suppress0", []> { // CHECK-NOT: static void build(::mlir::OpBuilder &odsBuilder, ::mlir::OperationState &odsState, ::mlir::TypeRange b, ::mlir::ValueRange a); // CHECK-NOT: static HCollectiveParamsSuppress0Op create(::mlir::OpBuilder &builder, ::mlir::Location location, ::mlir::TypeRange b, ::mlir::ValueRange a); // CHECK-NOT: static HCollectiveParamsSuppress0Op create(::mlir::ImplicitLocOpBuilder &builder, ::mlir::TypeRange b, ::mlir::ValueRange a); +// CHECK-NOT: use the overload taking Properties and discardableAttributes instead // CHECK: static void build(::mlir::OpBuilder &, ::mlir::OperationState &odsState, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, ::llvm::ArrayRef<::mlir::NamedAttribute> attributes = {}); // CHECK: static HCollectiveParamsSuppress0Op create(::mlir::OpBuilder &builder, ::mlir::Location location, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, ::llvm::ArrayRef<::mlir::NamedAttribute> attributes = {}); // CHECK: static HCollectiveParamsSuppress0Op create(::mlir::ImplicitLocOpBuilder &builder, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, ::llvm::ArrayRef<::mlir::NamedAttribute> attributes = {}); @@ -445,21 +489,24 @@ def NS_LOp : NS_Op<"op_with_same_operands_and_result_types_unwrapped_attr", [Sam // CHECK: static LOp create(::mlir::OpBuilder &builder, ::mlir::Location location, ::mlir::TypeRange resultTypes, ::mlir::Value a, ::mlir::Value b, uint32_t attr1); // CHECK: static LOp create(::mlir::ImplicitLocOpBuilder &builder, ::mlir::TypeRange resultTypes, ::mlir::Value a, ::mlir::Value b, uint32_t attr1); -// CHECK: static void build(::mlir::OpBuilder &, ::mlir::OperationState &odsState, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, ::llvm::ArrayRef<::mlir::NamedAttribute> attributes = {}); -// CHECK: static LOp create(::mlir::OpBuilder &builder, ::mlir::Location location, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, ::llvm::ArrayRef<::mlir::NamedAttribute> attributes = {}); -// CHECK: static LOp create(::mlir::ImplicitLocOpBuilder &builder, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, ::llvm::ArrayRef<::mlir::NamedAttribute> attributes = {}); - -// CHECK: static void build(::mlir::OpBuilder &odsBuilder, ::mlir::OperationState &odsState, ::mlir::ValueRange operands, ::llvm::ArrayRef<::mlir::NamedAttribute> attributes = {}); -// CHECK: static LOp create(::mlir::OpBuilder &builder, ::mlir::Location location, ::mlir::ValueRange operands, ::llvm::ArrayRef<::mlir::NamedAttribute> attributes = {}); -// CHECK: static LOp create(::mlir::ImplicitLocOpBuilder &builder, ::mlir::ValueRange operands, ::llvm::ArrayRef<::mlir::NamedAttribute> attributes = {}); - -// CHECK: static void build(::mlir::OpBuilder &, ::mlir::OperationState &odsState, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, const Properties &properties, ::llvm::ArrayRef<::mlir::NamedAttribute> discardableAttributes = {}); -// CHECK: static LOp create(::mlir::OpBuilder &builder, ::mlir::Location location, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, const Properties &properties, ::llvm::ArrayRef<::mlir::NamedAttribute> discardableAttributes = {}); -// CHECK: static LOp create(::mlir::ImplicitLocOpBuilder &builder, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, const Properties &properties, ::llvm::ArrayRef<::mlir::NamedAttribute> discardableAttributes = {}); - -// CHECK: static void build(::mlir::OpBuilder &odsBuilder, ::mlir::OperationState &odsState, ::mlir::ValueRange operands, const Properties &properties, ::llvm::ArrayRef<::mlir::NamedAttribute> discardableAttributes = {}); -// CHECK: static LOp create(::mlir::OpBuilder &builder, ::mlir::Location location, ::mlir::ValueRange operands, const Properties &properties, ::llvm::ArrayRef<::mlir::NamedAttribute> discardableAttributes = {}); -// CHECK: static LOp create(::mlir::ImplicitLocOpBuilder &builder, ::mlir::ValueRange operands, const Properties &properties, ::llvm::ArrayRef<::mlir::NamedAttribute> discardableAttributes = {}); +// CHECK{LITERAL}: [[deprecated("use the overload taking Properties and discardableAttributes instead")]] +// CHECK-NEXT: static void build(::mlir::OpBuilder &, ::mlir::OperationState &odsState, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, ::llvm::ArrayRef<::mlir::NamedAttribute> attributes = {}); +// CHECK{LITERAL}: [[deprecated("use the overload taking Properties and discardableAttributes instead")]] +// CHECK-NEXT: static LOp create(::mlir::OpBuilder &builder, ::mlir::Location location, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, ::llvm::ArrayRef<::mlir::NamedAttribute> attributes = {}); +// CHECK{LITERAL}: [[deprecated("use the overload taking Properties and discardableAttributes instead")]] +// CHECK-NEXT: static LOp create(::mlir::ImplicitLocOpBuilder &builder, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, ::llvm::ArrayRef<::mlir::NamedAttribute> attributes = {}); +// CHECK{LITERAL}: [[deprecated("use the overload taking Properties and discardableAttributes instead")]] +// CHECK-NEXT: static void build(::mlir::OpBuilder &odsBuilder, ::mlir::OperationState &odsState, ::mlir::ValueRange operands, ::llvm::ArrayRef<::mlir::NamedAttribute> attributes = {}); +// CHECK{LITERAL}: [[deprecated("use the overload taking Properties and discardableAttributes instead")]] +// CHECK-NEXT: static LOp create(::mlir::OpBuilder &builder, ::mlir::Location location, ::mlir::ValueRange operands, ::llvm::ArrayRef<::mlir::NamedAttribute> attributes = {}); +// CHECK{LITERAL}: [[deprecated("use the overload taking Properties and discardableAttributes instead")]] +// CHECK-NEXT: static LOp create(::mlir::ImplicitLocOpBuilder &builder, ::mlir::ValueRange operands, ::llvm::ArrayRef<::mlir::NamedAttribute> attributes = {}); +// CHECK-NEXT: static void build(::mlir::OpBuilder &, ::mlir::OperationState &odsState, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, const Properties &properties, ::llvm::ArrayRef<::mlir::NamedAttribute> discardableAttributes = {}); +// CHECK-NEXT: static LOp create(::mlir::OpBuilder &builder, ::mlir::Location location, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, const Properties &properties, ::llvm::ArrayRef<::mlir::NamedAttribute> discardableAttributes = {}); +// CHECK-NEXT: static LOp create(::mlir::ImplicitLocOpBuilder &builder, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, const Properties &properties, ::llvm::ArrayRef<::mlir::NamedAttribute> discardableAttributes = {}); +// CHECK-NEXT: static void build(::mlir::OpBuilder &odsBuilder, ::mlir::OperationState &odsState, ::mlir::ValueRange operands, const Properties &properties, ::llvm::ArrayRef<::mlir::NamedAttribute> discardableAttributes = {}); +// CHECK-NEXT: static LOp create(::mlir::OpBuilder &builder, ::mlir::Location location, ::mlir::ValueRange operands, const Properties &properties, ::llvm::ArrayRef<::mlir::NamedAttribute> discardableAttributes = {}); +// CHECK-NEXT: static LOp create(::mlir::ImplicitLocOpBuilder &builder, ::mlir::ValueRange operands, const Properties &properties, ::llvm::ArrayRef<::mlir::NamedAttribute> discardableAttributes = {}); def NS_MOp : NS_Op<"op_with_single_result_and_fold_adaptor_fold", []> { let results = (outs AnyType:$res); @@ -551,3 +598,22 @@ def _TypeInferredPropOp : NS_Op<"type_inferred_prop_op_with_properties", [ let results = (outs AnyType:$result); let hasCustomAssemblyFormat = 1; } + +// CHECK{LITERAL}: [[deprecated("use the overload taking Properties and discardableAttributes instead")]] +// CHECK-NEXT: static void build(::mlir::OpBuilder &, ::mlir::OperationState &odsState, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, ::llvm::ArrayRef<::mlir::NamedAttribute> attributes = {}); +// CHECK{LITERAL}: [[deprecated("use the overload taking Properties and discardableAttributes instead")]] +// CHECK-NEXT: static _TypeInferredPropOp create(::mlir::OpBuilder &builder, ::mlir::Location location, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, ::llvm::ArrayRef<::mlir::NamedAttribute> attributes = {}); +// CHECK{LITERAL}: [[deprecated("use the overload taking Properties and discardableAttributes instead")]] +// CHECK-NEXT: static _TypeInferredPropOp create(::mlir::ImplicitLocOpBuilder &builder, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, ::llvm::ArrayRef<::mlir::NamedAttribute> attributes = {}); +// CHECK{LITERAL}: [[deprecated("use the overload taking Properties and discardableAttributes instead")]] +// CHECK-NEXT: static void build(::mlir::OpBuilder &odsBuilder, ::mlir::OperationState &odsState, ::mlir::ValueRange operands, ::llvm::ArrayRef<::mlir::NamedAttribute> attributes = {}); +// CHECK{LITERAL}: [[deprecated("use the overload taking Properties and discardableAttributes instead")]] +// CHECK-NEXT: static _TypeInferredPropOp create(::mlir::OpBuilder &builder, ::mlir::Location location, ::mlir::ValueRange operands, ::llvm::ArrayRef<::mlir::NamedAttribute> attributes = {}); +// CHECK{LITERAL}: [[deprecated("use the overload taking Properties and discardableAttributes instead")]] +// CHECK-NEXT: static _TypeInferredPropOp create(::mlir::ImplicitLocOpBuilder &builder, ::mlir::ValueRange operands, ::llvm::ArrayRef<::mlir::NamedAttribute> attributes = {}); +// CHECK-NEXT: static void build(::mlir::OpBuilder &, ::mlir::OperationState &odsState, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, const Properties &properties, ::llvm::ArrayRef<::mlir::NamedAttribute> discardableAttributes = {}); +// CHECK-NEXT: static _TypeInferredPropOp create(::mlir::OpBuilder &builder, ::mlir::Location location, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, const Properties &properties, ::llvm::ArrayRef<::mlir::NamedAttribute> discardableAttributes = {}); +// CHECK-NEXT: static _TypeInferredPropOp create(::mlir::ImplicitLocOpBuilder &builder, ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands, const Properties &properties, ::llvm::ArrayRef<::mlir::NamedAttribute> discardableAttributes = {}); +// CHECK-NEXT: static void build(::mlir::OpBuilder &odsBuilder, ::mlir::OperationState &odsState, ::mlir::ValueRange operands, const Properties &properties, ::llvm::ArrayRef<::mlir::NamedAttribute> discardableAttributes = {}); +// CHECK-NEXT: static _TypeInferredPropOp create(::mlir::OpBuilder &builder, ::mlir::Location location, ::mlir::ValueRange operands, const Properties &properties, ::llvm::ArrayRef<::mlir::NamedAttribute> discardableAttributes = {}); +// CHECK-NEXT: static _TypeInferredPropOp create(::mlir::ImplicitLocOpBuilder &builder, ::mlir::ValueRange operands, const Properties &properties, ::llvm::ArrayRef<::mlir::NamedAttribute> discardableAttributes = {}); diff --git a/examples/mlir/_tblgen-regress/op-error.td b/examples/mlir/_tblgen-regress/op-error.td index a2eab1f..ec90891 100644 --- a/examples/mlir/_tblgen-regress/op-error.td +++ b/examples/mlir/_tblgen-regress/op-error.td @@ -11,6 +11,10 @@ // RUN: not mlir-tblgen -gen-op-decls -I %S/../../include -DERROR11 %s 2>&1 | FileCheck --check-prefix=ERROR11 %s // RUN: not mlir-tblgen -gen-op-decls -I %S/../../include -DERROR12 %s 2>&1 | FileCheck --check-prefix=ERROR12 %s // RUN: not mlir-tblgen -gen-op-decls -I %S/../../include -DERROR13 %s 2>&1 | FileCheck --check-prefix=ERROR13 %s +// RUN: not mlir-tblgen -gen-op-decls -I %S/../../include -DERROR14 %s 2>&1 | FileCheck --check-prefix=ERROR14 %s +// RUN: not mlir-tblgen -gen-op-decls -I %S/../../include -DERROR15 %s 2>&1 | FileCheck --check-prefix=ERROR15 %s +// RUN: not mlir-tblgen -gen-op-decls -I %S/../../include -DERROR16 %s 2>&1 | FileCheck --check-prefix=ERROR16 %s +// RUN: not mlir-tblgen -gen-op-decls -I %S/../../include -DERROR17 %s 2>&1 | FileCheck --check-prefix=ERROR17 %s include "mlir/IR/OpBase.td" @@ -124,3 +128,41 @@ def OpInterfaceB : OpInterface<"OpInterfaceB"> { // ERROR13: error: OpInterfaceB::Trait requires OpTraitA to precede it in traits list def OpInterfaceWithoutDependentTrait : Op {} #endif + +#ifdef ERROR14 +// ERROR14: error: property 'operandSegmentSizes' conflicts with the property generated by 'AttrSizedOperandSegments' +def OpWithOperandSegmentProperty : Op { + let arguments = (ins Variadic:$lhs, Variadic:$rhs, + I32Prop:$operandSegmentSizes); +} +#endif + +#ifdef ERROR15 +// ERROR15: error: property 'operand_segment_sizes' conflicts with the property generated by 'AttrSizedOperandSegments' +def OpWithLegacyOperandSegmentProperty + : Op { + let arguments = (ins Variadic:$lhs, Variadic:$rhs, + I32Prop:$operand_segment_sizes); +} +#endif + +#ifdef ERROR16 +// ERROR16: error: property 'resultSegmentSizes' conflicts with the property generated by 'AttrSizedResultSegments' +def OpWithResultSegmentProperty : Op { + let arguments = (ins I32Prop:$resultSegmentSizes); + let results = (outs Variadic:$lhs, Variadic:$rhs); +} +#endif + +#ifdef ERROR17 +// ERROR17: error: property 'result_segment_sizes' conflicts with the property generated by 'AttrSizedResultSegments' +def OpWithLegacyResultSegmentProperty + : Op { + let arguments = (ins I32Prop:$result_segment_sizes); + let results = (outs Variadic:$lhs, Variadic:$rhs); +} +#endif diff --git a/examples/mlir/_tblgen-regress/op-format-custom-properties-printer.td b/examples/mlir/_tblgen-regress/op-format-custom-properties-printer.td new file mode 100644 index 0000000..0919185 --- /dev/null +++ b/examples/mlir/_tblgen-regress/op-format-custom-properties-printer.td @@ -0,0 +1,32 @@ +// RUN: mlir-tblgen -gen-op-defs -I %S/../../include %s | FileCheck %s + +include "mlir/IR/OpBase.td" + +def TestDialect : Dialect { + let name = "test"; + let cppNamespace = "::test"; +} + +// A custom properties printer shadows the generated per-field helper. +// CHECK-NOT: CustomPropertiesPrinterOp::_odsPrintPropertiesAsKeyValueList +// CHECK: void CustomPropertiesPrinterOp::print +// CHECK-NOT: CustomPropertiesPrinterOp::_odsPrintPropertiesAsKeyValueList +def CustomPropertiesPrinterOp : Op { + let arguments = (ins I64Attr:$attr); + let assemblyFormat = "prop-dict attr-dict"; + let hasCustomPropertiesPrinter = 1; + let extraClassDeclaration = [{ + void printProperties(::mlir::MLIRContext *, ::mlir::OpAsmPrinter &, + const Properties &, + ::mlir::ArrayRef<::llvm::StringRef>); + }]; +} + +// Absent default-valued attributes are not printed from property storage. +// CHECK: void DefaultValuedAttrOp::_odsPrintPropertiesAsKeyValueList +// CHECK: if (shouldPrint("attr") && prop.attr) { +// CHECK-NEXT: printKey("attr", /*printValue=*/true); +def DefaultValuedAttrOp : Op { + let arguments = (ins DefaultValuedAttr:$attr); + let assemblyFormat = "prop-dict attr-dict"; +} diff --git a/examples/mlir/_tblgen-regress/op-format-spec.td b/examples/mlir/_tblgen-regress/op-format-spec.td index 0ad6d09..fac4386 100644 --- a/examples/mlir/_tblgen-regress/op-format-spec.td +++ b/examples/mlir/_tblgen-regress/op-format-spec.td @@ -152,6 +152,10 @@ def OIListCustom : TestFormat_Op<[{ | `nowait` | `reduction` custom($arg1, type($arg1))) attr-dict }], [AttrSizedOperandSegments]>, Arguments<(ins Optional:$arg0, Optional:$arg1)>; +def OIListWithSeparator : TestFormat_Op<[{ + oilist<`,`>( `keyword` $arg0 `:` type($arg0) + | `otherkeyword` $arg1 `:` type($arg1)) attr-dict +}], [AttrSizedOperandSegments]>, Arguments<(ins Optional:$arg0, Optional:$arg1)>; //===----------------------------------------------------------------------===// // Optional Groups diff --git a/examples/mlir/_tblgen-regress/op-format.td b/examples/mlir/_tblgen-regress/op-format.td index 1790737..f7f0c1b 100644 --- a/examples/mlir/_tblgen-regress/op-format.td +++ b/examples/mlir/_tblgen-regress/op-format.td @@ -1,14 +1,24 @@ // RUN: mlir-tblgen -gen-op-defs -I %S/../../include %s | FileCheck %s +// RUN: mlir-tblgen -gen-op-defs -I %S/../../include %s | FileCheck %s --check-prefix=PROP-ENUM include "mlir/IR/OpBase.td" +include "mlir/IR/EnumAttr.td" def TestDialect : Dialect { let name = "test"; } +def TestStrictPropertiesDialect : Dialect { + let name = "test_strict_properties"; + let useStrictPropertiesInAssemblyFormat = 1; +} class TestFormat_Op traits = []> : Op { let assemblyFormat = fmt; } +class TestStrictPropertiesFormat_Op traits = []> + : Op { + let assemblyFormat = fmt; +} //===----------------------------------------------------------------------===// // Directives @@ -18,6 +28,59 @@ class TestFormat_Op traits = []> // custom //===----------------------------------------------------------------------===// +// CHECK-LABEL: AttrDictContextOperand::parse +// CHECK: // coverity[ARRAY_VS_SINGLETON] +// CHECK-NEXT: if (parser.resolveOperands( +// CHECK-LABEL: AttrDictContextOperand::print +// CHECK: _odsPrinter.printOptionalAttrDict(_odsAttrs.getDictionary((*this)->getContext()).getValue(), elidedAttrs); +def AttrDictContextOperand : TestFormat_Op<[{ + $context `:` type($context) attr-dict +}]>, Arguments<(ins AnyType:$context)>; + +// CHECK-LABEL: AttrDictDefaultInherentAttr::parse +// CHECK: verifyInherentAttrs +def AttrDictDefaultInherentAttr : TestFormat_Op<[{ + $attr attr-dict +}]>, Arguments<(ins I64Attr:$attr)>; + +// CHECK-LABEL: AttrDictStrictInferredSegmentAttr::parse +// CHECK: result.getOrAddProperties().segment_sizes = parser.getBuilder().getDenseI32ArrayAttr(inputsOperandGroupSizes); +// CHECK-LABEL: AttrDictStrictInferredSegmentAttr::print +// CHECK: elidedAttrs.push_back("segment_sizes"); +def AttrDictStrictInferredSegmentAttr : TestStrictPropertiesFormat_Op<[{ + custom($inputs, type($inputs)) attr-dict +}]>, Arguments<(ins VariadicOfVariadic:$inputs, + DenseI32ArrayAttr:$segment_sizes)>; + +def PropDictStrictInferredSegmentAttr : TestStrictPropertiesFormat_Op<[{ + custom($inputs, type($inputs)) prop-dict attr-dict +}]>, Arguments<(ins VariadicOfVariadic:$inputs, + DenseI32ArrayAttr:$segment_sizes)>; + +// CHECK-LABEL: AttrDictStrictInherentAttr::parse +// CHECK-NOT: verifyInherentAttrs +// CHECK: if (result.attributes.get("attr")) +// CHECK-NEXT: return parser.emitError(loc, "inherent attribute 'attr' cannot +// CHECK-SAME: be parsed from attr-dict when strict properties in assembly +// CHECK-SAME: format is enabled"); +// CHECK-NOT: verifyInherentAttrs +// CHECK: return ::mlir::success(); +def AttrDictStrictInherentAttr : TestStrictPropertiesFormat_Op<[{ + $attr attr-dict +}]>, Arguments<(ins I64Attr:$attr)>; + +def AttrDictStrictProperty : TestStrictPropertiesFormat_Op<[{ + $prop attr-dict +}]>, Arguments<(ins IntProp<"int64_t">:$prop)>; + +def AttrDictStrictPropDict : TestStrictPropertiesFormat_Op<[{ + prop-dict attr-dict +}]>, Arguments<(ins IntProp<"int64_t">:$prop)>; + +def AttrDictStrictPropDictInherentAttr : TestStrictPropertiesFormat_Op<[{ + prop-dict attr-dict +}]>, Arguments<(ins I64Attr:$attr)>; + // CHECK-LABEL: CustomStringLiteralA::parse // CHECK: parseFoo({{.*}}, parser.getBuilder().getI1Type()) // CHECK-LABEL: CustomStringLiteralA::print @@ -50,6 +113,212 @@ def CustomStringLiteralD : TestFormat_Op<[{ custom(prop-dict) attr-dict }]>; +// A directive keyword can still name a custom directive. +def CustomEnumKeyword : TestFormat_Op<[{ + custom($attr) attr-dict +}]>, Arguments<(ins I64Attr:$attr)>; + +//===----------------------------------------------------------------------===// +// EnumAttr formatting +//===----------------------------------------------------------------------===// + +// Test that EnumAttr (backed by EnumInfo, not EnumAttrInfo) is recognized as +// an enum attribute and uses the enum-keyword format path. + +def TestEnumCase0 : I32EnumCase<"Case0", 0>; +def TestEnumCase1 : I32EnumCase<"Case1", 1>; + +def TestEnum : I32Enum<"TestEnum", "a test enum", [TestEnumCase0, TestEnumCase1]> { + let cppNamespace = "::test"; +} + +def TestEnumAttr : EnumAttr; + +def TestPrettyEnumAttr : EnumAttr; + +def TestCustomEnum + : I32Enum<"TestCustomEnum", "an enum with custom attribute syntax", [ + I32EnumCase<"Case0", 0>, + I32EnumCase<"Case1", 1>, + ]> { + let cppNamespace = "::test"; +} +def TestCustomEnumAttr + : EnumAttr { + let assemblyFormat = "custom($value)"; +} + +def TestNonKeywordEnumCase : I32EnumCase<"NonKeyword", 0, "non-keyword">; +def TestNonKeywordEnum + : I32Enum<"TestNonKeywordEnum", "a non-keyword enum", + [TestNonKeywordEnumCase]> { + let cppNamespace = "::test"; +} +def TestNonKeywordEnumAttr + : EnumAttr; + +// Unquoted bit enums use a separator-aware attribute parser instead of the +// operation-level enum parser, which only accepts one keyword or string. + +def TestBitEnumNone : I32BitEnumCaseNone<"None">; +def TestBitEnumBit0 : I32BitEnumCaseBit<"Bit0", 0>; +def TestBitEnumBit1 : I32BitEnumCaseBit<"Bit1", 1>; + +def TestBitEnum : I32BitEnum<"TestBitEnum", "a test bit enum", + [TestBitEnumNone, TestBitEnumBit0, + TestBitEnumBit1]> { + let cppNamespace = "::test"; + let printBitEnumQuoted = 0; +} + +def TestBitEnumAttr : EnumAttr; + +def TestCommaBitEnum + : I32BitEnum<"TestCommaBitEnum", "a comma-separated bit enum", [ + TestBitEnumNone, + I32BitEnumCaseBit<"Bit0", 0>, + I32BitEnumCaseBit<"Bit1", 1>, + ]> { + let cppNamespace = "::test"; + let printBitEnumQuoted = 0; + let separator = ","; +} +def TestCommaBitEnumAttr + : EnumAttr; + +// Comma-separated bit enums use square brackets to disambiguate their commas +// from separators between prop-dict fields. + +// PROP-ENUM-LABEL: CommaBitEnumAttrPropDictOp::parsePropertiesFromKeyValueList +// PROP-ENUM: parser.parseLSquare() +// PROP-ENUM: symbolizeTestCommaBitEnum +// PROP-ENUM: parser.parseOptionalComma() +// PROP-ENUM: parser.parseRSquare() +// PROP-ENUM-LABEL: CommaBitEnumAttrPropDictOp::_odsPrintPropertiesAsKeyValueList +// PROP-ENUM: _odsPrinter << "[" +// PROP-ENUM: stringifyTestCommaBitEnum +// PROP-ENUM: _odsPrinter << "]" +def CommaBitEnumAttrPropDictOp : TestFormat_Op<"prop-dict attr-dict">, + Arguments<(ins TestCommaBitEnumAttr:$flags, I64Attr:$next)>; + +// Custom EnumAttr formats continue to use their own parser and printer. + +// PROP-ENUM-LABEL: CustomEnumAttrPropDictOp::parsePropertiesFromKeyValueList +// PROP-ENUM: parseCustomAttributeWithFallback +// PROP-ENUM-LABEL: CustomEnumAttrPropDictOp::_odsPrintPropertiesAsKeyValueList +// PROP-ENUM: printStrippedAttrOrType +def CustomEnumAttrPropDictOp : TestFormat_Op<"prop-dict attr-dict">, + Arguments<(ins TestCustomEnumAttr:$attr)>; + +// Keyed prop-dict fields strip the default EnumAttr body syntax. + +// PROP-ENUM-LABEL: DefaultEnumAttrPropDictOp::parsePropertiesFromKeyValueList +// PROP-ENUM: symbolizeTestEnum +// PROP-ENUM-LABEL: DefaultEnumAttrPropDictOp::_odsPrintPropertiesAsKeyValueList +// PROP-ENUM: stringifyTestEnum +def DefaultEnumAttrPropDictOp : TestFormat_Op<"prop-dict attr-dict">, + Arguments<(ins TestEnumAttr:$attr)>; + +// Default-valued optional attributes have a non-optional getter. + +// CHECK-LABEL: DefaultOptionalEnumAttrOp::print +// CHECK: auto caseValue = getAttr(); +def DefaultOptionalEnumAttrOp : TestFormat_Op<"(enum($attr)^)? attr-dict">, + Arguments<(ins DefaultValuedOptionalAttr< + TestEnumAttr, "::test::TestEnum::Case0">:$attr)>; + +// CHECK-LABEL: EnumAttrOp::parse +// CHECK: symbolizeTestEnum +// CHECK-LABEL: EnumAttrOp::print +// CHECK: stringifyTestEnum +def EnumAttrOp : TestFormat_Op<"enum($attr) attr-dict">, + Arguments<(ins TestEnumAttr:$attr)>; + +// CHECK-LABEL: EnumAttrUnquotedBitOp::parse +// CHECK: parseCustomAttributeWithFallback +// CHECK-NOT: symbolizeTestBitEnum +// CHECK-LABEL: EnumAttrUnquotedBitOp::print +// CHECK: printStrippedAttrOrType(getAttrAttr()) +def EnumAttrUnquotedBitOp : TestFormat_Op<"$attr attr-dict">, + Arguments<(ins TestBitEnumAttr:$attr)>; + +// Non-keyword enum values use the quoted-string fallback. + +// CHECK-LABEL: EnumDirectiveNonKeywordOp::parse +// CHECK: parser.parseOptionalKeyword(&attrStr, {}) +// CHECK: parser.parseOptionalAttribute +// CHECK-LABEL: EnumDirectiveNonKeywordOp::print +// CHECK: _odsPrinter << '"' << caseValueStr << '"' +def EnumDirectiveNonKeywordOp : TestFormat_Op<"enum($attr) attr-dict">, + Arguments<(ins TestNonKeywordEnumAttr:$attr)>; + +// The enum directive formats the symbolic enum value independently of the +// attribute's custom assembly format. + +// CHECK-LABEL: EnumDirectiveOp::parse +// CHECK: symbolizeTestEnum +// CHECK-NOT: parseCustomAttributeWithFallback +// CHECK-LABEL: EnumDirectiveOp::print +// CHECK: stringifyTestEnum +// CHECK-NOT: printStrippedAttrOrType +def EnumDirectiveOp : TestFormat_Op<"enum($attr) attr-dict">, + Arguments<(ins TestPrettyEnumAttr:$attr)>; + +// The enum directive preserves the separator-aware parser and unquoted printer +// for an unquoted bit enum. + +// CHECK-LABEL: EnumDirectiveUnquotedBitOp::parse +// CHECK: symbolizeTestBitEnum +// CHECK: parseOptionalVerticalBar +// CHECK-LABEL: EnumDirectiveUnquotedBitOp::print +// CHECK: stringifyTestBitEnum +// CHECK: _odsPrinter << caseValueStr +def EnumDirectiveUnquotedBitOp : TestFormat_Op<"enum($attr) attr-dict">, + Arguments<(ins TestBitEnumAttr:$attr)>; + +// Test that legacy EnumAttrInfo-based attributes (I32EnumAttr) also use the +// enum-keyword format path. + +def LegacyEnumCase0 : I32EnumAttrCase<"LCase0", 0>; +def LegacyEnumCase1 : I32EnumAttrCase<"LCase1", 1>; + +def LegacyTestEnum : I32EnumAttr<"LegacyTestEnum", "a legacy test enum", + [LegacyEnumCase0, LegacyEnumCase1]> { + let cppNamespace = "::test"; +} + +// CHECK-LABEL: LegacyEnumAttrOp::parse +// CHECK: symbolizeLegacyTestEnum +// CHECK-LABEL: LegacyEnumAttrOp::print +// CHECK: stringifyLegacyTestEnum +def LegacyEnumAttrOp : TestFormat_Op<"$attr attr-dict">, + Arguments<(ins LegacyTestEnum:$attr)>; + +// Explicit enum formatting must not change the existing parser and printer +// selected for legacy EnumAttrInfo attributes written as plain variables. + +def TestLegacyBitEnumNone : I32BitEnumAttrCaseNone<"None", "none">; +def TestLegacyBitEnumBit0 : I32BitEnumAttrCaseBit<"Bit0", 0, "bit0">; +def TestLegacyBitEnumBit1 : I32BitEnumAttrCaseBit<"Bit1", 1, "bit1">; +def TestLegacyUnquotedBitEnum + : I32BitEnumAttr<"TestLegacyUnquotedBitEnum", "a legacy bit enum", + [TestLegacyBitEnumNone, TestLegacyBitEnumBit0, + TestLegacyBitEnumBit1]> { + let cppNamespace = "::test"; + let printBitEnumQuoted = 0; +} + +// CHECK-LABEL: LegacyUnquotedBitEnumOp::parse +// CHECK: parser.parseOptionalKeyword +// CHECK-NOT: parseOptionalVerticalBar +// CHECK-NOT: parseOptionalComma +// CHECK-LABEL: LegacyUnquotedBitEnumOp::print +// CHECK: switch (caseValue) +// CHECK: default: +// CHECK-NEXT: _odsPrinter << '"' << caseValueStr << '"' +def LegacyUnquotedBitEnumOp : TestFormat_Op<"$attr attr-dict">, + Arguments<(ins TestLegacyUnquotedBitEnum:$attr)>; + //===----------------------------------------------------------------------===// // Optional Groups //===----------------------------------------------------------------------===// @@ -88,10 +357,24 @@ def OptionalGroupB : TestFormat_Op<[{ // CHECK-NEXT: odsPrinter << ' '; // CHECK-NEXT: odsPrinter.printAttributeWithoutType(getAAttr()); // CHECK-NEXT: } +// CHECK: elidedAttrs.push_back("a"); +// CHECK-NOT: elidedAttrs.push_back("a"); +// CHECK: _odsPrinter.printOptionalAttrDict def OptionalGroupC : TestFormat_Op<[{ ($a^)? attr-dict }]>, Arguments<(ins DefaultValuedStrAttr:$a)>; +// CHECK-LABEL: OptionalGroupCPropDict::print +// CHECK: elidedProps.push_back("a"); +// CHECK-NOT: elidedProps.push_back("a"); +// CHECK: printProperties +// CHECK: elidedAttrs.push_back("a"); +// CHECK-NOT: elidedAttrs.push_back("a"); +// CHECK: _odsPrinter.printOptionalAttrDict +def OptionalGroupCPropDict : TestFormat_Op<[{ + ($a^)? prop-dict attr-dict +}]>, Arguments<(ins DefaultValuedStrAttr:$a)>; + // CHECK-LABEL: OptionalGroupD::parse // CHECK: if (auto optResult = [&]() -> ::mlir::OptionalParseResult { // CHECK: auto odsResult = parseCustom(parser, aOperand, bOperand); @@ -110,6 +393,42 @@ def OptionalGroupD : TestFormat_Op<[{ (custom($a, $b)^)? attr-dict }], [AttrSizedOperandSegments]>, Arguments<(ins Optional:$a, Optional:$b)>; +//===----------------------------------------------------------------------===// +// prop-dict + AttrSizedOperandSegments +//===----------------------------------------------------------------------===// + +// When the format spells out each variadic operand group individually, the +// segment sizes can be inferred from what was actually parsed, so the +// `operandSegmentSizes` key is left completely untouched: it isn't read, +// validated, or marked as a used key at all. +// CHECK-LABEL: PropDictSegmentSizesInferred::setPropertiesFromParsedAttr +// CHECK-NOT: operandSegmentSizes +// CHECK: for (::mlir::NamedAttribute attr : dict) { +def PropDictSegmentSizesInferred : TestFormat_Op<[{ + $a1 `:` $a2 prop-dict attr-dict +}], [AttrSizedOperandSegments]>, + Arguments<(ins Variadic:$a1, Variadic:$a2)>; + +// When the format uses a bulk `operands`/`type(operands)` directive, the +// segment sizes can't be reconstructed from the parse, so the +// `operandSegmentSizes` key is required and validated. +// CHECK-LABEL: PropDictSegmentSizesRequired::setPropertiesFromParsedAttr +// CHECK: auto operandSegmentSizesAttrName = ::mlir::StringAttr::get(ctx, "operandSegmentSizes"); +// CHECK-NEXT: usedKeys.insert(operandSegmentSizesAttrName); +// CHECK-NEXT: auto attr = dict.get(operandSegmentSizesAttrName); +// CHECK-NEXT: if (!attr) { +// CHECK: if (::mlir::failed(::mlir::convertFromAttribute(prop.operandSegmentSizes, attr, +def PropDictSegmentSizesRequired : TestFormat_Op<[{ + `(` operands `)` `:` type(operands) prop-dict attr-dict +}], [AttrSizedOperandSegments]>, + Arguments<(ins Variadic:$a1, Variadic:$a2)>; + +// CHECK-LABEL: PropDictStrictInferredSegmentAttr::setPropertiesFromParsedAttr +// CHECK-NOT: segment_sizes +// CHECK: return ::mlir::success(); +// CHECK-LABEL: PropDictStrictInferredSegmentAttr::print +// CHECK: elidedProps.push_back("segment_sizes"); + // CHECK-LABEL: RegionRef::parse // CHECK: auto odsResult = parseCustom(parser, *bodyRegion); // CHECK-LABEL: RegionRef::print diff --git a/examples/mlir/_tblgen-regress/op-result.td b/examples/mlir/_tblgen-regress/op-result.td index a4f7af6..ddc1ef0 100644 --- a/examples/mlir/_tblgen-regress/op-result.td +++ b/examples/mlir/_tblgen-regress/op-result.td @@ -14,6 +14,9 @@ def OpA : NS_Op<"one_normal_result_op", []> { let results = (outs I32:$result); } +// DECL-LABEL: class OpA : {{.*}} { +// DECL: static constexpr int odsIndex_result = 0; + // CHECK-LABEL: void OpA::build // CHECK: ::mlir::TypeRange resultTypes, ::mlir::ValueRange operands // CHECK: assert(resultTypes.size() == 1u && "mismatched number of return types"); @@ -24,6 +27,10 @@ def OpB : NS_Op<"same_input_output_type_op", [SameOperandsAndResultType]> { let results = (outs I32:$y); } +// DECL-LABEL: class OpB : {{.*}} { +// DECL: static constexpr int odsIndex_x = 0; +// DECL: static constexpr int odsIndex_y = 0; + // CHECK-LABEL: OpB definitions // CHECK: void OpB::build(::mlir::OpBuilder &odsBuilder, ::mlir::OperationState &odsState, ::mlir::Type y, ::mlir::Value x) // CHECK: odsState.addTypes(y); @@ -39,6 +46,10 @@ def OpC : NS_Op<"three_normal_result_op", []> { let results = (outs I32:$x, /*unnamed*/I32, I32:$z); } +// DECL-LABEL: class OpC : {{.*}} { +// DECL: static constexpr int odsIndex_x = 0; +// DECL: static constexpr int odsIndex_z = 2; + // CHECK-LABEL: OpC definitions // CHECK: void OpC::build(::mlir::OpBuilder &odsBuilder, ::mlir::OperationState &odsState, ::mlir::Type x, ::mlir::Type resultType1, ::mlir::Type z) // CHECK-NEXT: odsState.addTypes(x) diff --git a/examples/mlir/_tblgen-regress/op-side-effects.td b/examples/mlir/_tblgen-regress/op-side-effects.td index f08b373..ca24c7f 100644 --- a/examples/mlir/_tblgen-regress/op-side-effects.td +++ b/examples/mlir/_tblgen-regress/op-side-effects.td @@ -9,6 +9,8 @@ class TEST_Op traits = []> : Op; def CustomResource : Resource<"CustomResource">; +def ParameterizedResource : Resource<"CustomResource", + [{::mlir::StringAttr::get(getContext(), "parameter")}]>; def SideEffectOpA : TEST_Op<"side_effect_op_a"> { let arguments = (ins @@ -24,6 +26,13 @@ def SideEffectOpA : TEST_Op<"side_effect_op_a"> { def SideEffectOpB : TEST_Op<"side_effect_op_b", [MemoryEffects<[MemWrite]>]>; +def SideEffectOpC : TEST_Op<"side_effect_op_c"> { + let arguments = (ins Arg]>); +} + +def SideEffectOpD : TEST_Op<"side_effect_op_d", + [MemoryEffects<[MemWrite]>]>; + // CHECK: void SideEffectOpA::getEffects // CHECK: { // CHECK: auto valueRange = getODSOperandIndexAndLength(0); @@ -50,3 +59,9 @@ def SideEffectOpB : TEST_Op<"side_effect_op_b", // CHECK: void SideEffectOpB::getEffects // CHECK: effects.emplace_back(::mlir::MemoryEffects::Write::get(), 0, false, CustomResource::get()); + +// CHECK-LABEL: void SideEffectOpC::getEffects +// CHECK: "parameter"), 0, false, CustomResource::get()); + +// CHECK-LABEL: void SideEffectOpD::getEffects +// CHECK: "parameter"), 0, false, CustomResource::get()); diff --git a/examples/mlir/_tblgen-regress/predicate.td b/examples/mlir/_tblgen-regress/predicate.td index 41e041f..ae43688 100644 --- a/examples/mlir/_tblgen-regress/predicate.td +++ b/examples/mlir/_tblgen-regress/predicate.td @@ -27,9 +27,9 @@ def OpA : NS_Op<"op_for_CPred_containing_multiple_same_placeholder", []> { // CHECK-NOT. << " must be 32-bit integer or floating-point type, but got " << type; // CHECK: static ::llvm::LogicalResult [[$TENSOR_CONSTRAINT:__mlir_ods_local_type_constraint.*]]( -// CHECK: if (!(((::llvm::isa<::mlir::TensorType>(type))) && ([](::mlir::Type elementType) { return (true); }(::llvm::cast<::mlir::ShapedType>(type).getElementType())))) { +// CHECK: if (!(((::llvm::isa<::mlir::TensorType>(type))) && ([](::mlir::Type elementType) { return !((::llvm::isa<::mlir::TokenType>(elementType))); }(::llvm::cast<::mlir::ShapedType>(type).getElementType())))) { // CHECK-NEXT: return op->emitOpError(valueKind) << " #" << valueIndex -// CHECK-NEXT: << " must be tensor of any type values, but got " << type; +// CHECK-NEXT: << " must be tensor of any non-token type values, but got " << type; // CHECK: static ::llvm::LogicalResult [[$TENSOR_INTEGER_FLOAT_CONSTRAINT:__mlir_ods_local_type_constraint.*]]( // CHECK: if (!(((::llvm::isa<::mlir::TensorType>(type))) && ([](::mlir::Type elementType) { return ((elementType.isF32())) || ((elementType.isSignlessInteger(32))); }(::llvm::cast<::mlir::ShapedType>(type).getElementType())))) {