IntrinsicsAMDGPU.td 80 KB

12345678910111213141516171819202122232425262728293031323334353637383940414243444546474849505152535455565758596061626364656667686970717273747576777879808182838485868788899091929394959697989910010110210310410510610710810911011111211311411511611711811912012112212312412512612712812913013113213313413513613713813914014114214314414514614714814915015115215315415515615715815916016116216316416516616716816917017117217317417517617717817918018118218318418518618718818919019119219319419519619719819920020120220320420520620720820921021121221321421521621721821922022122222322422522622722822923023123223323423523623723823924024124224324424524624724824925025125225325425525625725825926026126226326426526626726826927027127227327427527627727827928028128228328428528628728828929029129229329429529629729829930030130230330430530630730830931031131231331431531631731831932032132232332432532632732832933033133233333433533633733833934034134234334434534634734834935035135235335435535635735835936036136236336436536636736836937037137237337437537637737837938038138238338438538638738838939039139239339439539639739839940040140240340440540640740840941041141241341441541641741841942042142242342442542642742842943043143243343443543643743843944044144244344444544644744844945045145245345445545645745845946046146246346446546646746846947047147247347447547647747847948048148248348448548648748848949049149249349449549649749849950050150250350450550650750850951051151251351451551651751851952052152252352452552652752852953053153253353453553653753853954054154254354454554654754854955055155255355455555655755855956056156256356456556656756856957057157257357457557657757857958058158258358458558658758858959059159259359459559659759859960060160260360460560660760860961061161261361461561661761861962062162262362462562662762862963063163263363463563663763863964064164264364464564664764864965065165265365465565665765865966066166266366466566666766866967067167267367467567667767867968068168268368468568668768868969069169269369469569669769869970070170270370470570670770870971071171271371471571671771871972072172272372472572672772872973073173273373473573673773873974074174274374474574674774874975075175275375475575675775875976076176276376476576676776876977077177277377477577677777877978078178278378478578678778878979079179279379479579679779879980080180280380480580680780880981081181281381481581681781881982082182282382482582682782882983083183283383483583683783883984084184284384484584684784884985085185285385485585685785885986086186286386486586686786886987087187287387487587687787887988088188288388488588688788888989089189289389489589689789889990090190290390490590690790890991091191291391491591691791891992092192292392492592692792892993093193293393493593693793893994094194294394494594694794894995095195295395495595695795895996096196296396496596696796896997097197297397497597697797897998098198298398498598698798898999099199299399499599699799899910001001100210031004100510061007100810091010101110121013101410151016101710181019102010211022102310241025102610271028102910301031103210331034103510361037103810391040104110421043104410451046104710481049105010511052105310541055105610571058105910601061106210631064106510661067106810691070107110721073107410751076107710781079108010811082108310841085108610871088108910901091109210931094109510961097109810991100110111021103110411051106110711081109111011111112111311141115111611171118111911201121112211231124112511261127112811291130113111321133113411351136113711381139114011411142114311441145114611471148114911501151115211531154115511561157115811591160116111621163116411651166116711681169117011711172117311741175117611771178117911801181118211831184118511861187118811891190119111921193119411951196119711981199120012011202120312041205120612071208120912101211121212131214121512161217121812191220122112221223122412251226122712281229123012311232123312341235123612371238123912401241124212431244124512461247124812491250125112521253125412551256125712581259126012611262126312641265126612671268126912701271127212731274127512761277127812791280128112821283128412851286128712881289129012911292129312941295129612971298129913001301130213031304130513061307130813091310131113121313131413151316131713181319132013211322132313241325132613271328132913301331133213331334133513361337133813391340134113421343134413451346134713481349135013511352135313541355135613571358135913601361136213631364136513661367136813691370137113721373137413751376137713781379138013811382138313841385138613871388138913901391139213931394139513961397139813991400140114021403140414051406140714081409141014111412141314141415141614171418141914201421142214231424142514261427142814291430143114321433143414351436143714381439144014411442144314441445144614471448144914501451145214531454145514561457145814591460146114621463146414651466146714681469147014711472147314741475147614771478147914801481148214831484148514861487148814891490149114921493149414951496149714981499150015011502150315041505150615071508150915101511151215131514151515161517151815191520152115221523152415251526152715281529153015311532153315341535153615371538153915401541154215431544154515461547154815491550155115521553155415551556155715581559156015611562156315641565156615671568156915701571157215731574157515761577157815791580158115821583158415851586158715881589159015911592159315941595159615971598159916001601160216031604160516061607160816091610161116121613161416151616161716181619162016211622162316241625162616271628162916301631163216331634163516361637163816391640164116421643164416451646164716481649165016511652165316541655165616571658165916601661166216631664166516661667166816691670167116721673167416751676167716781679168016811682168316841685168616871688168916901691169216931694169516961697169816991700170117021703170417051706170717081709171017111712171317141715171617171718171917201721172217231724172517261727172817291730173117321733173417351736173717381739174017411742174317441745174617471748174917501751175217531754175517561757175817591760176117621763176417651766176717681769177017711772177317741775177617771778177917801781178217831784178517861787178817891790179117921793179417951796179717981799180018011802180318041805180618071808180918101811181218131814181518161817181818191820182118221823182418251826182718281829183018311832183318341835183618371838183918401841184218431844184518461847184818491850185118521853185418551856185718581859186018611862186318641865186618671868186918701871187218731874187518761877187818791880188118821883188418851886188718881889189018911892189318941895189618971898189919001901190219031904190519061907190819091910191119121913191419151916191719181919192019211922192319241925192619271928192919301931193219331934193519361937193819391940194119421943194419451946194719481949195019511952195319541955195619571958195919601961196219631964196519661967196819691970197119721973197419751976197719781979
  1. //===- IntrinsicsAMDGPU.td - Defines AMDGPU intrinsics -----*- tablegen -*-===//
  2. //
  3. // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
  4. // See https://llvm.org/LICENSE.txt for license information.
  5. // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
  6. //
  7. //===----------------------------------------------------------------------===//
  8. //
  9. // This file defines all of the R600-specific intrinsics.
  10. //
  11. //===----------------------------------------------------------------------===//
  12. class AMDGPUReadPreloadRegisterIntrinsic
  13. : Intrinsic<[llvm_i32_ty], [], [IntrNoMem, IntrSpeculatable, IntrWillReturn]>;
  14. class AMDGPUReadPreloadRegisterIntrinsicNamed<string name>
  15. : Intrinsic<[llvm_i32_ty], [], [IntrNoMem, IntrSpeculatable, IntrWillReturn]>, GCCBuiltin<name>;
  16. // Used to tag image and resource intrinsics with information used to generate
  17. // mem operands.
  18. class AMDGPURsrcIntrinsic<int rsrcarg, bit isimage = false> {
  19. int RsrcArg = rsrcarg;
  20. bit IsImage = isimage;
  21. }
  22. let TargetPrefix = "r600" in {
  23. multiclass AMDGPUReadPreloadRegisterIntrinsic_xyz {
  24. def _x : AMDGPUReadPreloadRegisterIntrinsic;
  25. def _y : AMDGPUReadPreloadRegisterIntrinsic;
  26. def _z : AMDGPUReadPreloadRegisterIntrinsic;
  27. }
  28. multiclass AMDGPUReadPreloadRegisterIntrinsic_xyz_named<string prefix> {
  29. def _x : AMDGPUReadPreloadRegisterIntrinsicNamed<!strconcat(prefix, "_x")>;
  30. def _y : AMDGPUReadPreloadRegisterIntrinsicNamed<!strconcat(prefix, "_y")>;
  31. def _z : AMDGPUReadPreloadRegisterIntrinsicNamed<!strconcat(prefix, "_z")>;
  32. }
  33. defm int_r600_read_global_size : AMDGPUReadPreloadRegisterIntrinsic_xyz_named
  34. <"__builtin_r600_read_global_size">;
  35. defm int_r600_read_ngroups : AMDGPUReadPreloadRegisterIntrinsic_xyz_named
  36. <"__builtin_r600_read_ngroups">;
  37. defm int_r600_read_tgid : AMDGPUReadPreloadRegisterIntrinsic_xyz_named
  38. <"__builtin_r600_read_tgid">;
  39. defm int_r600_read_local_size : AMDGPUReadPreloadRegisterIntrinsic_xyz;
  40. defm int_r600_read_tidig : AMDGPUReadPreloadRegisterIntrinsic_xyz;
  41. def int_r600_group_barrier : GCCBuiltin<"__builtin_r600_group_barrier">,
  42. Intrinsic<[], [], [IntrConvergent, IntrWillReturn]>;
  43. // AS 7 is PARAM_I_ADDRESS, used for kernel arguments
  44. def int_r600_implicitarg_ptr :
  45. GCCBuiltin<"__builtin_r600_implicitarg_ptr">,
  46. Intrinsic<[LLVMQualPointerType<llvm_i8_ty, 7>], [],
  47. [IntrNoMem, IntrSpeculatable, IntrWillReturn]>;
  48. def int_r600_rat_store_typed :
  49. // 1st parameter: Data
  50. // 2nd parameter: Index
  51. // 3rd parameter: Constant RAT ID
  52. Intrinsic<[], [llvm_v4i32_ty, llvm_v4i32_ty, llvm_i32_ty], [IntrWillReturn]>,
  53. GCCBuiltin<"__builtin_r600_rat_store_typed">;
  54. def int_r600_recipsqrt_ieee : Intrinsic<
  55. [llvm_anyfloat_ty], [LLVMMatchType<0>], [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  56. >;
  57. def int_r600_recipsqrt_clamped : Intrinsic<
  58. [llvm_anyfloat_ty], [LLVMMatchType<0>], [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  59. >;
  60. def int_r600_cube : Intrinsic<
  61. [llvm_v4f32_ty], [llvm_v4f32_ty], [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  62. >;
  63. def int_r600_store_stream_output : Intrinsic<
  64. [], [llvm_v4f32_ty, llvm_i32_ty, llvm_i32_ty, llvm_i32_ty], [IntrWillReturn]
  65. >;
  66. class TextureIntrinsicFloatInput : Intrinsic<[llvm_v4f32_ty], [
  67. llvm_v4f32_ty, // Coord
  68. llvm_i32_ty, // offset_x
  69. llvm_i32_ty, // offset_y,
  70. llvm_i32_ty, // offset_z,
  71. llvm_i32_ty, // resource_id
  72. llvm_i32_ty, // samplerid
  73. llvm_i32_ty, // coord_type_x
  74. llvm_i32_ty, // coord_type_y
  75. llvm_i32_ty, // coord_type_z
  76. llvm_i32_ty], // coord_type_w
  77. [IntrNoMem, IntrWillReturn]
  78. >;
  79. class TextureIntrinsicInt32Input : Intrinsic<[llvm_v4i32_ty], [
  80. llvm_v4i32_ty, // Coord
  81. llvm_i32_ty, // offset_x
  82. llvm_i32_ty, // offset_y,
  83. llvm_i32_ty, // offset_z,
  84. llvm_i32_ty, // resource_id
  85. llvm_i32_ty, // samplerid
  86. llvm_i32_ty, // coord_type_x
  87. llvm_i32_ty, // coord_type_y
  88. llvm_i32_ty, // coord_type_z
  89. llvm_i32_ty], // coord_type_w
  90. [IntrNoMem, IntrWillReturn]
  91. >;
  92. def int_r600_store_swizzle :
  93. Intrinsic<[], [llvm_v4f32_ty, llvm_i32_ty, llvm_i32_ty], [IntrWillReturn]
  94. >;
  95. def int_r600_tex : TextureIntrinsicFloatInput;
  96. def int_r600_texc : TextureIntrinsicFloatInput;
  97. def int_r600_txl : TextureIntrinsicFloatInput;
  98. def int_r600_txlc : TextureIntrinsicFloatInput;
  99. def int_r600_txb : TextureIntrinsicFloatInput;
  100. def int_r600_txbc : TextureIntrinsicFloatInput;
  101. def int_r600_txf : TextureIntrinsicInt32Input;
  102. def int_r600_txq : TextureIntrinsicInt32Input;
  103. def int_r600_ddx : TextureIntrinsicFloatInput;
  104. def int_r600_ddy : TextureIntrinsicFloatInput;
  105. def int_r600_dot4 : Intrinsic<[llvm_float_ty],
  106. [llvm_v4f32_ty, llvm_v4f32_ty], [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  107. >;
  108. def int_r600_kill : Intrinsic<[], [llvm_float_ty], [IntrWillReturn]>;
  109. } // End TargetPrefix = "r600"
  110. let TargetPrefix = "amdgcn" in {
  111. //===----------------------------------------------------------------------===//
  112. // ABI Special Intrinsics
  113. //===----------------------------------------------------------------------===//
  114. defm int_amdgcn_workitem_id : AMDGPUReadPreloadRegisterIntrinsic_xyz;
  115. defm int_amdgcn_workgroup_id : AMDGPUReadPreloadRegisterIntrinsic_xyz_named
  116. <"__builtin_amdgcn_workgroup_id">;
  117. def int_amdgcn_dispatch_ptr :
  118. Intrinsic<[LLVMQualPointerType<llvm_i8_ty, 4>], [],
  119. [Align<RetIndex, 4>, IntrNoMem, IntrSpeculatable, IntrWillReturn]>;
  120. def int_amdgcn_queue_ptr :
  121. GCCBuiltin<"__builtin_amdgcn_queue_ptr">,
  122. Intrinsic<[LLVMQualPointerType<llvm_i8_ty, 4>], [],
  123. [Align<RetIndex, 4>, IntrNoMem, IntrSpeculatable, IntrWillReturn]>;
  124. def int_amdgcn_kernarg_segment_ptr :
  125. GCCBuiltin<"__builtin_amdgcn_kernarg_segment_ptr">,
  126. Intrinsic<[LLVMQualPointerType<llvm_i8_ty, 4>], [],
  127. [Align<RetIndex, 4>, IntrNoMem, IntrSpeculatable, IntrWillReturn]>;
  128. def int_amdgcn_implicitarg_ptr :
  129. GCCBuiltin<"__builtin_amdgcn_implicitarg_ptr">,
  130. Intrinsic<[LLVMQualPointerType<llvm_i8_ty, 4>], [],
  131. [Align<RetIndex, 4>, IntrNoMem, IntrSpeculatable, IntrWillReturn]>;
  132. def int_amdgcn_groupstaticsize :
  133. GCCBuiltin<"__builtin_amdgcn_groupstaticsize">,
  134. Intrinsic<[llvm_i32_ty], [], [IntrNoMem, IntrSpeculatable, IntrWillReturn]>;
  135. def int_amdgcn_dispatch_id :
  136. GCCBuiltin<"__builtin_amdgcn_dispatch_id">,
  137. Intrinsic<[llvm_i64_ty], [], [IntrNoMem, IntrSpeculatable, IntrWillReturn]>;
  138. def int_amdgcn_implicit_buffer_ptr :
  139. GCCBuiltin<"__builtin_amdgcn_implicit_buffer_ptr">,
  140. Intrinsic<[LLVMQualPointerType<llvm_i8_ty, 4>], [],
  141. [Align<RetIndex, 4>, IntrNoMem, IntrSpeculatable, IntrWillReturn]>;
  142. // Set EXEC to the 64-bit value given.
  143. // This is always moved to the beginning of the basic block.
  144. // FIXME: Should be mangled for wave size.
  145. def int_amdgcn_init_exec : Intrinsic<[],
  146. [llvm_i64_ty], // 64-bit literal constant
  147. [IntrConvergent, ImmArg<ArgIndex<0>>]>;
  148. // Set EXEC according to a thread count packed in an SGPR input:
  149. // thread_count = (input >> bitoffset) & 0x7f;
  150. // This is always moved to the beginning of the basic block.
  151. // Note: only inreg arguments to the parent function are valid as
  152. // inputs to this intrinsic, computed values cannot be used.
  153. def int_amdgcn_init_exec_from_input : Intrinsic<[],
  154. [llvm_i32_ty, // 32-bit SGPR input
  155. llvm_i32_ty], // bit offset of the thread count
  156. [IntrConvergent, ImmArg<ArgIndex<1>>]>;
  157. def int_amdgcn_wavefrontsize :
  158. GCCBuiltin<"__builtin_amdgcn_wavefrontsize">,
  159. Intrinsic<[llvm_i32_ty], [], [IntrNoMem, IntrSpeculatable, IntrWillReturn]>;
  160. //===----------------------------------------------------------------------===//
  161. // Instruction Intrinsics
  162. //===----------------------------------------------------------------------===//
  163. // The first parameter is s_sendmsg immediate (i16),
  164. // the second one is copied to m0
  165. def int_amdgcn_s_sendmsg : GCCBuiltin<"__builtin_amdgcn_s_sendmsg">,
  166. Intrinsic <[], [llvm_i32_ty, llvm_i32_ty],
  167. [ImmArg<ArgIndex<0>>, IntrNoMem, IntrHasSideEffects]>;
  168. def int_amdgcn_s_sendmsghalt : GCCBuiltin<"__builtin_amdgcn_s_sendmsghalt">,
  169. Intrinsic <[], [llvm_i32_ty, llvm_i32_ty],
  170. [ImmArg<ArgIndex<0>>, IntrNoMem, IntrHasSideEffects]>;
  171. def int_amdgcn_s_barrier : GCCBuiltin<"__builtin_amdgcn_s_barrier">,
  172. Intrinsic<[], [], [IntrNoMem, IntrHasSideEffects, IntrConvergent, IntrWillReturn]>;
  173. def int_amdgcn_wave_barrier : GCCBuiltin<"__builtin_amdgcn_wave_barrier">,
  174. Intrinsic<[], [], [IntrNoMem, IntrHasSideEffects, IntrConvergent, IntrWillReturn]>;
  175. def int_amdgcn_s_waitcnt : GCCBuiltin<"__builtin_amdgcn_s_waitcnt">,
  176. Intrinsic<[], [llvm_i32_ty], [ImmArg<ArgIndex<0>>, IntrNoMem, IntrHasSideEffects, IntrWillReturn]>;
  177. def int_amdgcn_div_scale : Intrinsic<
  178. // 1st parameter: Numerator
  179. // 2nd parameter: Denominator
  180. // 3rd parameter: Select quotient. Must equal Numerator or Denominator.
  181. // (0 = Denominator, 1 = Numerator).
  182. [llvm_anyfloat_ty, llvm_i1_ty],
  183. [LLVMMatchType<0>, LLVMMatchType<0>, llvm_i1_ty],
  184. [IntrNoMem, IntrSpeculatable, ImmArg<ArgIndex<2>>, IntrWillReturn]
  185. >;
  186. def int_amdgcn_div_fmas : Intrinsic<[llvm_anyfloat_ty],
  187. [LLVMMatchType<0>, LLVMMatchType<0>, LLVMMatchType<0>, llvm_i1_ty],
  188. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  189. >;
  190. def int_amdgcn_div_fixup : Intrinsic<[llvm_anyfloat_ty],
  191. [LLVMMatchType<0>, LLVMMatchType<0>, LLVMMatchType<0>],
  192. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  193. >;
  194. // Look Up 2.0 / pi src0 with segment select src1[4:0]
  195. def int_amdgcn_trig_preop : Intrinsic<
  196. [llvm_anyfloat_ty], [LLVMMatchType<0>, llvm_i32_ty],
  197. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  198. >;
  199. def int_amdgcn_sin : Intrinsic<
  200. [llvm_anyfloat_ty], [LLVMMatchType<0>],
  201. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  202. >;
  203. def int_amdgcn_cos : Intrinsic<
  204. [llvm_anyfloat_ty], [LLVMMatchType<0>], [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  205. >;
  206. def int_amdgcn_log_clamp : Intrinsic<
  207. [llvm_anyfloat_ty], [LLVMMatchType<0>], [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  208. >;
  209. def int_amdgcn_fmul_legacy : GCCBuiltin<"__builtin_amdgcn_fmul_legacy">,
  210. Intrinsic<[llvm_float_ty], [llvm_float_ty, llvm_float_ty],
  211. [IntrNoMem, IntrSpeculatable, IntrWillReturn, Commutative]
  212. >;
  213. // Fused single-precision multiply-add with legacy behaviour for the multiply,
  214. // which is that +/- 0.0 * anything (even NaN or infinity) is +0.0. This is
  215. // intended for use on subtargets that have the v_fma_legacy_f32 and/or
  216. // v_fmac_legacy_f32 instructions. (Note that v_fma_legacy_f16 is unrelated and
  217. // has a completely different kind of legacy behaviour.)
  218. def int_amdgcn_fma_legacy :
  219. Intrinsic<[llvm_float_ty], [llvm_float_ty, llvm_float_ty, llvm_float_ty],
  220. [IntrNoMem, IntrSpeculatable, IntrWillReturn, Commutative]
  221. >;
  222. def int_amdgcn_rcp : Intrinsic<
  223. [llvm_anyfloat_ty], [LLVMMatchType<0>], [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  224. >;
  225. def int_amdgcn_rcp_legacy : GCCBuiltin<"__builtin_amdgcn_rcp_legacy">,
  226. Intrinsic<[llvm_float_ty], [llvm_float_ty],
  227. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  228. >;
  229. def int_amdgcn_sqrt : Intrinsic<
  230. [llvm_anyfloat_ty], [LLVMMatchType<0>], [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  231. >;
  232. def int_amdgcn_rsq : Intrinsic<
  233. [llvm_anyfloat_ty], [LLVMMatchType<0>], [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  234. >;
  235. def int_amdgcn_rsq_legacy : GCCBuiltin<"__builtin_amdgcn_rsq_legacy">,
  236. Intrinsic<
  237. [llvm_float_ty], [llvm_float_ty], [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  238. >;
  239. // out = 1.0 / sqrt(a) result clamped to +/- max_float.
  240. def int_amdgcn_rsq_clamp : Intrinsic<
  241. [llvm_anyfloat_ty], [LLVMMatchType<0>], [IntrNoMem, IntrSpeculatable, IntrWillReturn]>;
  242. def int_amdgcn_ldexp : Intrinsic<
  243. [llvm_anyfloat_ty], [LLVMMatchType<0>, llvm_i32_ty],
  244. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  245. >;
  246. def int_amdgcn_frexp_mant : Intrinsic<
  247. [llvm_anyfloat_ty], [LLVMMatchType<0>], [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  248. >;
  249. def int_amdgcn_frexp_exp : Intrinsic<
  250. [llvm_anyint_ty], [llvm_anyfloat_ty], [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  251. >;
  252. // v_fract is buggy on SI/CI. It mishandles infinities, may return 1.0
  253. // and always uses rtz, so is not suitable for implementing the OpenCL
  254. // fract function. It should be ok on VI.
  255. def int_amdgcn_fract : Intrinsic<
  256. [llvm_anyfloat_ty], [LLVMMatchType<0>], [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  257. >;
  258. def int_amdgcn_cvt_pkrtz : GCCBuiltin<"__builtin_amdgcn_cvt_pkrtz">,
  259. Intrinsic<[llvm_v2f16_ty], [llvm_float_ty, llvm_float_ty],
  260. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  261. >;
  262. def int_amdgcn_cvt_pknorm_i16 :
  263. GCCBuiltin<"__builtin_amdgcn_cvt_pknorm_i16">,
  264. Intrinsic<[llvm_v2i16_ty], [llvm_float_ty, llvm_float_ty],
  265. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  266. >;
  267. def int_amdgcn_cvt_pknorm_u16 :
  268. GCCBuiltin<"__builtin_amdgcn_cvt_pknorm_u16">,
  269. Intrinsic<[llvm_v2i16_ty], [llvm_float_ty, llvm_float_ty],
  270. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  271. >;
  272. def int_amdgcn_cvt_pk_i16 :
  273. GCCBuiltin<"__builtin_amdgcn_cvt_pk_i16">,
  274. Intrinsic<
  275. [llvm_v2i16_ty], [llvm_i32_ty, llvm_i32_ty],
  276. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  277. >;
  278. def int_amdgcn_cvt_pk_u16 : GCCBuiltin<"__builtin_amdgcn_cvt_pk_u16">,
  279. Intrinsic<[llvm_v2i16_ty], [llvm_i32_ty, llvm_i32_ty],
  280. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  281. >;
  282. def int_amdgcn_class : Intrinsic<
  283. [llvm_i1_ty], [llvm_anyfloat_ty, llvm_i32_ty],
  284. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  285. >;
  286. def int_amdgcn_fmed3 : GCCBuiltin<"__builtin_amdgcn_fmed3">,
  287. Intrinsic<[llvm_anyfloat_ty],
  288. [LLVMMatchType<0>, LLVMMatchType<0>, LLVMMatchType<0>],
  289. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  290. >;
  291. def int_amdgcn_cubeid : GCCBuiltin<"__builtin_amdgcn_cubeid">,
  292. Intrinsic<[llvm_float_ty],
  293. [llvm_float_ty, llvm_float_ty, llvm_float_ty],
  294. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  295. >;
  296. def int_amdgcn_cubema : GCCBuiltin<"__builtin_amdgcn_cubema">,
  297. Intrinsic<[llvm_float_ty],
  298. [llvm_float_ty, llvm_float_ty, llvm_float_ty],
  299. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  300. >;
  301. def int_amdgcn_cubesc : GCCBuiltin<"__builtin_amdgcn_cubesc">,
  302. Intrinsic<[llvm_float_ty],
  303. [llvm_float_ty, llvm_float_ty, llvm_float_ty],
  304. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  305. >;
  306. def int_amdgcn_cubetc : GCCBuiltin<"__builtin_amdgcn_cubetc">,
  307. Intrinsic<[llvm_float_ty],
  308. [llvm_float_ty, llvm_float_ty, llvm_float_ty],
  309. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  310. >;
  311. // v_ffbh_i32, as opposed to v_ffbh_u32. For v_ffbh_u32, llvm.ctlz
  312. // should be used.
  313. def int_amdgcn_sffbh :
  314. Intrinsic<[llvm_anyint_ty], [LLVMMatchType<0>],
  315. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  316. >;
  317. // v_mad_f32|f16/v_mac_f32|f16, selected regardless of denorm support.
  318. def int_amdgcn_fmad_ftz :
  319. Intrinsic<[llvm_anyfloat_ty],
  320. [LLVMMatchType<0>, LLVMMatchType<0>, LLVMMatchType<0>],
  321. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  322. >;
  323. // Fields should mirror atomicrmw
  324. class AMDGPUAtomicIncIntrin : Intrinsic<[llvm_anyint_ty],
  325. [llvm_anyptr_ty,
  326. LLVMMatchType<0>,
  327. llvm_i32_ty, // ordering
  328. llvm_i32_ty, // scope
  329. llvm_i1_ty], // isVolatile
  330. [IntrArgMemOnly, IntrWillReturn, NoCapture<ArgIndex<0>>,
  331. ImmArg<ArgIndex<2>>, ImmArg<ArgIndex<3>>, ImmArg<ArgIndex<4>>], "",
  332. [SDNPMemOperand]
  333. >;
  334. def int_amdgcn_atomic_inc : AMDGPUAtomicIncIntrin;
  335. def int_amdgcn_atomic_dec : AMDGPUAtomicIncIntrin;
  336. class AMDGPULDSIntrin :
  337. Intrinsic<[llvm_any_ty],
  338. [LLVMQualPointerType<LLVMMatchType<0>, 3>,
  339. LLVMMatchType<0>,
  340. llvm_i32_ty, // ordering
  341. llvm_i32_ty, // scope
  342. llvm_i1_ty], // isVolatile
  343. [IntrArgMemOnly, IntrWillReturn, NoCapture<ArgIndex<0>>,
  344. ImmArg<ArgIndex<2>>, ImmArg<ArgIndex<3>>, ImmArg<ArgIndex<4>>]
  345. >;
  346. // FIXME: The m0 argument should be moved after the normal arguments
  347. class AMDGPUDSOrderedIntrinsic : Intrinsic<
  348. [llvm_i32_ty],
  349. // M0 = {hi16:address, lo16:waveID}. Allow passing M0 as a pointer, so that
  350. // the bit packing can be optimized at the IR level.
  351. [LLVMQualPointerType<llvm_i32_ty, 2>, // IntToPtr(M0)
  352. llvm_i32_ty, // value to add or swap
  353. llvm_i32_ty, // ordering
  354. llvm_i32_ty, // scope
  355. llvm_i1_ty, // isVolatile
  356. llvm_i32_ty, // ordered count index (OA index), also added to the address
  357. // gfx10: bits 24-27 indicate the number of active threads/dwords
  358. llvm_i1_ty, // wave release, usually set to 1
  359. llvm_i1_ty], // wave done, set to 1 for the last ordered instruction
  360. [IntrWillReturn, NoCapture<ArgIndex<0>>,
  361. ImmArg<ArgIndex<2>>, ImmArg<ArgIndex<3>>, ImmArg<ArgIndex<4>>,
  362. ImmArg<ArgIndex<5>>, ImmArg<ArgIndex<6>>, ImmArg<ArgIndex<7>>
  363. ]
  364. >;
  365. class AMDGPUDSAppendConsumedIntrinsic : Intrinsic<
  366. [llvm_i32_ty],
  367. [llvm_anyptr_ty, // LDS or GDS ptr
  368. llvm_i1_ty], // isVolatile
  369. [IntrConvergent, IntrWillReturn, IntrArgMemOnly,
  370. NoCapture<ArgIndex<0>>, ImmArg<ArgIndex<1>>],
  371. "",
  372. [SDNPMemOperand]
  373. >;
  374. def int_amdgcn_ds_ordered_add : AMDGPUDSOrderedIntrinsic;
  375. def int_amdgcn_ds_ordered_swap : AMDGPUDSOrderedIntrinsic;
  376. // The pointer argument is assumed to be dynamically uniform if a VGPR.
  377. def int_amdgcn_ds_append : AMDGPUDSAppendConsumedIntrinsic;
  378. def int_amdgcn_ds_consume : AMDGPUDSAppendConsumedIntrinsic;
  379. def int_amdgcn_ds_fadd : AMDGPULDSIntrin;
  380. def int_amdgcn_ds_fmin : AMDGPULDSIntrin;
  381. def int_amdgcn_ds_fmax : AMDGPULDSIntrin;
  382. } // TargetPrefix = "amdgcn"
  383. // New-style image intrinsics
  384. //////////////////////////////////////////////////////////////////////////
  385. // Dimension-aware image intrinsics framework
  386. //////////////////////////////////////////////////////////////////////////
  387. // Helper class to represent (type, name) combinations of arguments. The
  388. // argument names are explanatory and used as DAG operand names for codegen
  389. // pattern matching.
  390. class AMDGPUArg<LLVMType ty, string name> {
  391. LLVMType Type = ty;
  392. string Name = name;
  393. }
  394. // Return [AMDGPUArg<basety, names[0]>, AMDGPUArg<LLVMMatchType<0>, names[1]>, ...]
  395. class makeArgList<list<string> names, LLVMType basety> {
  396. list<AMDGPUArg> ret =
  397. !listconcat([AMDGPUArg<basety, names[0]>],
  398. !foreach(name, !tail(names), AMDGPUArg<LLVMMatchType<0>, name>));
  399. }
  400. // Return arglist, with LLVMMatchType's references shifted by 'shift'.
  401. class arglistmatchshift<list<AMDGPUArg> arglist, int shift> {
  402. list<AMDGPUArg> ret =
  403. !foreach(arg, arglist,
  404. !if(!isa<LLVMMatchType>(arg.Type),
  405. AMDGPUArg<LLVMMatchType<!add(!cast<LLVMMatchType>(arg.Type).Number, shift)>,
  406. arg.Name>,
  407. arg));
  408. }
  409. // Return the concatenation of the given arglists. LLVMMatchType's are adjusted
  410. // accordingly, and shifted by an additional 'shift'.
  411. class arglistconcat<list<list<AMDGPUArg>> arglists, int shift = 0> {
  412. list<AMDGPUArg> ret =
  413. !foldl([]<AMDGPUArg>, arglists, lhs, rhs,
  414. !listconcat(
  415. lhs,
  416. arglistmatchshift<rhs,
  417. !add(shift, !foldl(0, lhs, a, b,
  418. !add(a, b.Type.isAny)))>.ret));
  419. }
  420. // Represent texture/image types / dimensionality.
  421. class AMDGPUDimProps<bits<3> enc, string name, string asmsuffix,
  422. list<string> coord_names, list<string> slice_names,
  423. bit msaa = 0> {
  424. AMDGPUDimProps Dim = !cast<AMDGPUDimProps>(NAME);
  425. string Name = name; // e.g. "2darraymsaa"
  426. string AsmSuffix = asmsuffix; // e.g. 2D_MSAA_ARRAY (used in assembly strings)
  427. bits<3> Encoding = enc;
  428. bit DA = 0; // DA bit in MIMG encoding
  429. bit MSAA = msaa;
  430. list<AMDGPUArg> CoordSliceArgs =
  431. makeArgList<!listconcat(coord_names, slice_names), llvm_anyfloat_ty>.ret;
  432. list<AMDGPUArg> CoordSliceIntArgs =
  433. makeArgList<!listconcat(coord_names, slice_names), llvm_anyint_ty>.ret;
  434. list<AMDGPUArg> GradientArgs =
  435. makeArgList<!listconcat(!foreach(name, coord_names, "d" # name # "dh"),
  436. !foreach(name, coord_names, "d" # name # "dv")),
  437. llvm_anyfloat_ty>.ret;
  438. bits<8> NumCoords = !size(CoordSliceArgs);
  439. bits<8> NumGradients = !size(GradientArgs);
  440. }
  441. def AMDGPUDim1D : AMDGPUDimProps<0x0, "1d", "1D", ["s"], []>;
  442. def AMDGPUDim2D : AMDGPUDimProps<0x1, "2d", "2D", ["s", "t"], []>;
  443. def AMDGPUDim3D : AMDGPUDimProps<0x2, "3d", "3D", ["s", "t", "r"], []>;
  444. let DA = 1 in {
  445. def AMDGPUDimCube : AMDGPUDimProps<0x3, "cube", "CUBE", ["s", "t"], ["face"]>;
  446. def AMDGPUDim1DArray : AMDGPUDimProps<0x4, "1darray", "1D_ARRAY", ["s"], ["slice"]>;
  447. def AMDGPUDim2DArray : AMDGPUDimProps<0x5, "2darray", "2D_ARRAY", ["s", "t"], ["slice"]>;
  448. }
  449. def AMDGPUDim2DMsaa : AMDGPUDimProps<0x6, "2dmsaa", "2D_MSAA", ["s", "t"], ["fragid"], 1>;
  450. let DA = 1 in {
  451. def AMDGPUDim2DArrayMsaa : AMDGPUDimProps<0x7, "2darraymsaa", "2D_MSAA_ARRAY", ["s", "t"], ["slice", "fragid"], 1>;
  452. }
  453. def AMDGPUDims {
  454. list<AMDGPUDimProps> NoMsaa = [AMDGPUDim1D, AMDGPUDim2D, AMDGPUDim3D,
  455. AMDGPUDimCube, AMDGPUDim1DArray,
  456. AMDGPUDim2DArray];
  457. list<AMDGPUDimProps> Msaa = [AMDGPUDim2DMsaa, AMDGPUDim2DArrayMsaa];
  458. list<AMDGPUDimProps> All = !listconcat(NoMsaa, Msaa);
  459. }
  460. // Represent sample variants, i.e. _C, _O, _B, ... and combinations thereof.
  461. class AMDGPUSampleVariant<string ucmod, string lcmod, list<AMDGPUArg> extra_addr> {
  462. string UpperCaseMod = ucmod;
  463. string LowerCaseMod = lcmod;
  464. // {offset} {bias} {z-compare}
  465. list<AMDGPUArg> ExtraAddrArgs = extra_addr;
  466. bit Gradients = false;
  467. // Name of the {lod} or {clamp} argument that is appended to the coordinates,
  468. // if any.
  469. string LodOrClamp = "";
  470. }
  471. // AMDGPUSampleVariants: all variants supported by IMAGE_SAMPLE
  472. // AMDGPUSampleVariantsNoGradients: variants supported by IMAGE_GATHER4
  473. defset list<AMDGPUSampleVariant> AMDGPUSampleVariants = {
  474. multiclass AMDGPUSampleHelper_Offset<string ucmod, string lcmod,
  475. list<AMDGPUArg> extra_addr> {
  476. def NAME#lcmod : AMDGPUSampleVariant<ucmod, lcmod, extra_addr>;
  477. def NAME#lcmod#_o : AMDGPUSampleVariant<
  478. ucmod#"_O", lcmod#"_o", !listconcat([AMDGPUArg<llvm_i32_ty, "offset">], extra_addr)>;
  479. }
  480. multiclass AMDGPUSampleHelper_Compare<string ucmod, string lcmod,
  481. list<AMDGPUArg> extra_addr> {
  482. defm NAME : AMDGPUSampleHelper_Offset<ucmod, lcmod, extra_addr>;
  483. defm NAME : AMDGPUSampleHelper_Offset<
  484. "_C"#ucmod, "_c"#lcmod, !listconcat(extra_addr, [AMDGPUArg<llvm_float_ty, "zcompare">])>;
  485. }
  486. multiclass AMDGPUSampleHelper_Clamp<string ucmod, string lcmod,
  487. list<AMDGPUArg> extra_addr> {
  488. defm NAME : AMDGPUSampleHelper_Compare<ucmod, lcmod, extra_addr>;
  489. let LodOrClamp = "clamp" in
  490. defm NAME : AMDGPUSampleHelper_Compare<ucmod#"_CL", lcmod#"_cl", extra_addr>;
  491. }
  492. defset list<AMDGPUSampleVariant> AMDGPUSampleVariantsNoGradients = {
  493. defm AMDGPUSample : AMDGPUSampleHelper_Clamp<"", "", []>;
  494. defm AMDGPUSample : AMDGPUSampleHelper_Clamp<
  495. "_B", "_b", [AMDGPUArg<llvm_anyfloat_ty, "bias">]>;
  496. let LodOrClamp = "lod" in
  497. defm AMDGPUSample : AMDGPUSampleHelper_Compare<"_L", "_l", []>;
  498. defm AMDGPUSample : AMDGPUSampleHelper_Compare<"_LZ", "_lz", []>;
  499. }
  500. let Gradients = true in {
  501. defm AMDGPUSample : AMDGPUSampleHelper_Clamp<"_D", "_d", []>;
  502. defm AMDGPUSample : AMDGPUSampleHelper_Clamp<"_CD", "_cd", []>;
  503. }
  504. }
  505. // Helper class to capture the profile of a dimension-aware image intrinsic.
  506. // This information is used to generate the intrinsic's type and to inform
  507. // codegen pattern matching.
  508. class AMDGPUDimProfile<string opmod,
  509. AMDGPUDimProps dim> {
  510. AMDGPUDimProps Dim = dim;
  511. string OpMod = opmod; // the corresponding instruction is named IMAGE_OpMod
  512. // These are intended to be overwritten by subclasses
  513. bit IsSample = false;
  514. bit IsAtomic = false;
  515. list<LLVMType> RetTypes = [];
  516. list<AMDGPUArg> DataArgs = [];
  517. list<AMDGPUArg> ExtraAddrArgs = [];
  518. bit Gradients = false;
  519. string LodClampMip = "";
  520. int NumRetAndDataAnyTypes =
  521. !foldl(0, !listconcat(RetTypes, !foreach(arg, DataArgs, arg.Type)), a, b,
  522. !add(a, b.isAny));
  523. list<AMDGPUArg> AddrArgs =
  524. arglistconcat<[ExtraAddrArgs,
  525. !if(Gradients, dim.GradientArgs, []),
  526. !listconcat(!if(IsSample, dim.CoordSliceArgs, dim.CoordSliceIntArgs),
  527. !if(!empty(LodClampMip),
  528. []<AMDGPUArg>,
  529. [AMDGPUArg<LLVMMatchType<0>, LodClampMip>]))],
  530. NumRetAndDataAnyTypes>.ret;
  531. list<LLVMType> AddrTypes = !foreach(arg, AddrArgs, arg.Type);
  532. list<AMDGPUArg> AddrDefaultArgs =
  533. !foreach(arg, AddrArgs,
  534. AMDGPUArg<!if(!or(arg.Type.isAny, !isa<LLVMMatchType>(arg.Type)),
  535. !if(IsSample, llvm_float_ty, llvm_i32_ty), arg.Type),
  536. arg.Name>);
  537. list<AMDGPUArg> AddrA16Args =
  538. !foreach(arg, AddrArgs,
  539. AMDGPUArg<!if(!or(arg.Type.isAny, !isa<LLVMMatchType>(arg.Type)),
  540. !if(IsSample, llvm_half_ty, llvm_i16_ty), arg.Type),
  541. arg.Name>);
  542. }
  543. class AMDGPUDimProfileCopy<AMDGPUDimProfile base> : AMDGPUDimProfile<base.OpMod, base.Dim> {
  544. let IsSample = base.IsSample;
  545. let IsAtomic = base.IsAtomic;
  546. let RetTypes = base.RetTypes;
  547. let DataArgs = base.DataArgs;
  548. let ExtraAddrArgs = base.ExtraAddrArgs;
  549. let Gradients = base.Gradients;
  550. let LodClampMip = base.LodClampMip;
  551. }
  552. class AMDGPUDimSampleProfile<string opmod,
  553. AMDGPUDimProps dim,
  554. AMDGPUSampleVariant sample> : AMDGPUDimProfile<opmod, dim> {
  555. let IsSample = true;
  556. let RetTypes = [llvm_any_ty];
  557. let ExtraAddrArgs = sample.ExtraAddrArgs;
  558. let Gradients = sample.Gradients;
  559. let LodClampMip = sample.LodOrClamp;
  560. }
  561. class AMDGPUDimNoSampleProfile<string opmod,
  562. AMDGPUDimProps dim,
  563. list<LLVMType> retty,
  564. list<AMDGPUArg> dataargs,
  565. bit Mip = false> : AMDGPUDimProfile<opmod, dim> {
  566. let RetTypes = retty;
  567. let DataArgs = dataargs;
  568. let LodClampMip = !if(Mip, "mip", "");
  569. }
  570. class AMDGPUDimAtomicProfile<string opmod,
  571. AMDGPUDimProps dim,
  572. list<AMDGPUArg> dataargs> : AMDGPUDimProfile<opmod, dim> {
  573. let RetTypes = [llvm_anyint_ty];
  574. let DataArgs = dataargs;
  575. let IsAtomic = true;
  576. }
  577. class AMDGPUDimGetResInfoProfile<AMDGPUDimProps dim> : AMDGPUDimProfile<"GET_RESINFO", dim> {
  578. let RetTypes = [llvm_anyfloat_ty];
  579. let DataArgs = [];
  580. let AddrArgs = [AMDGPUArg<llvm_anyint_ty, "mip">];
  581. let LodClampMip = "mip";
  582. }
  583. // Helper class for figuring out image intrinsic argument indexes.
  584. class AMDGPUImageDimIntrinsicEval<AMDGPUDimProfile P_> {
  585. int NumDataArgs = !size(P_.DataArgs);
  586. int NumDmaskArgs = !not(P_.IsAtomic);
  587. int NumExtraAddrArgs = !size(P_.ExtraAddrArgs);
  588. int NumVAddrArgs = !size(P_.AddrArgs);
  589. int NumGradientArgs = !if(P_.Gradients, !size(P_.Dim.GradientArgs), 0);
  590. int NumCoordArgs = !if(P_.IsSample, !size(P_.Dim.CoordSliceArgs), !size(P_.Dim.CoordSliceIntArgs));
  591. int NumRSrcArgs = 1;
  592. int NumSampArgs = !if(P_.IsSample, 2, 0);
  593. int DmaskArgIndex = NumDataArgs;
  594. int VAddrArgIndex = !add(DmaskArgIndex, NumDmaskArgs);
  595. int GradientArgIndex = !add(VAddrArgIndex, NumExtraAddrArgs);
  596. int CoordArgIndex = !add(GradientArgIndex, NumGradientArgs);
  597. int LodArgIndex = !add(VAddrArgIndex, NumVAddrArgs, -1);
  598. int MipArgIndex = LodArgIndex;
  599. int RsrcArgIndex = !add(VAddrArgIndex, NumVAddrArgs);
  600. int SampArgIndex = !add(RsrcArgIndex, NumRSrcArgs);
  601. int UnormArgIndex = !add(SampArgIndex, 1);
  602. int TexFailCtrlArgIndex = !add(SampArgIndex, NumSampArgs);
  603. int CachePolicyArgIndex = !add(TexFailCtrlArgIndex, 1);
  604. }
  605. // All dimension-aware intrinsics are derived from this class.
  606. class AMDGPUImageDimIntrinsic<AMDGPUDimProfile P_,
  607. list<IntrinsicProperty> props,
  608. list<SDNodeProperty> sdnodeprops> : Intrinsic<
  609. P_.RetTypes, // vdata(VGPR) -- for load/atomic-with-return
  610. !listconcat(
  611. !foreach(arg, P_.DataArgs, arg.Type), // vdata(VGPR) -- for store/atomic
  612. !if(P_.IsAtomic, [], [llvm_i32_ty]), // dmask(imm)
  613. P_.AddrTypes, // vaddr(VGPR)
  614. [llvm_v8i32_ty], // rsrc(SGPR)
  615. !if(P_.IsSample, [llvm_v4i32_ty, // samp(SGPR)
  616. llvm_i1_ty], []), // unorm(imm)
  617. [llvm_i32_ty, // texfailctrl(imm; bit 0 = tfe, bit 1 = lwe)
  618. llvm_i32_ty]), // cachepolicy(imm; bit 0 = glc, bit 1 = slc, bit 2 = dlc)
  619. !listconcat(props,
  620. !if(P_.IsAtomic, [], [ImmArg<ArgIndex<AMDGPUImageDimIntrinsicEval<P_>.DmaskArgIndex>>]),
  621. !if(P_.IsSample, [ImmArg<ArgIndex<AMDGPUImageDimIntrinsicEval<P_>.UnormArgIndex>>], []),
  622. [IntrWillReturn],
  623. [ImmArg<ArgIndex<AMDGPUImageDimIntrinsicEval<P_>.TexFailCtrlArgIndex>>,
  624. ImmArg<ArgIndex<AMDGPUImageDimIntrinsicEval<P_>.CachePolicyArgIndex>>]),
  625. "", sdnodeprops>,
  626. AMDGPURsrcIntrinsic<!add(!size(P_.DataArgs), !size(P_.AddrTypes),
  627. !if(P_.IsAtomic, 0, 1)), 1> {
  628. AMDGPUDimProfile P = P_;
  629. AMDGPUImageDimIntrinsic Intr = !cast<AMDGPUImageDimIntrinsic>(NAME);
  630. let TargetPrefix = "amdgcn";
  631. }
  632. // Marker class for intrinsics with a DMask that determines the returned
  633. // channels.
  634. class AMDGPUImageDMaskIntrinsic;
  635. defset list<AMDGPUImageDimIntrinsic> AMDGPUImageDimIntrinsics = {
  636. //////////////////////////////////////////////////////////////////////////
  637. // Load and store intrinsics
  638. //////////////////////////////////////////////////////////////////////////
  639. multiclass AMDGPUImageDimIntrinsicsNoMsaa<string opmod,
  640. list<LLVMType> retty,
  641. list<AMDGPUArg> dataargs,
  642. list<IntrinsicProperty> props,
  643. list<SDNodeProperty> sdnodeprops,
  644. bit Mip = false> {
  645. foreach dim = AMDGPUDims.NoMsaa in {
  646. def !strconcat(NAME, "_", dim.Name)
  647. : AMDGPUImageDimIntrinsic<
  648. AMDGPUDimNoSampleProfile<opmod, dim, retty, dataargs, Mip>,
  649. props, sdnodeprops>;
  650. }
  651. }
  652. multiclass AMDGPUImageDimIntrinsicsAll<string opmod,
  653. list<LLVMType> retty,
  654. list<AMDGPUArg> dataargs,
  655. list<IntrinsicProperty> props,
  656. list<SDNodeProperty> sdnodeprops,
  657. bit Mip = false> {
  658. foreach dim = AMDGPUDims.All in {
  659. def !strconcat(NAME, "_", dim.Name)
  660. : AMDGPUImageDimIntrinsic<
  661. AMDGPUDimNoSampleProfile<opmod, dim, retty, dataargs, Mip>,
  662. props, sdnodeprops>;
  663. }
  664. }
  665. defm int_amdgcn_image_load
  666. : AMDGPUImageDimIntrinsicsAll<"LOAD", [llvm_any_ty], [], [IntrReadMem],
  667. [SDNPMemOperand]>,
  668. AMDGPUImageDMaskIntrinsic;
  669. defm int_amdgcn_image_load_mip
  670. : AMDGPUImageDimIntrinsicsNoMsaa<"LOAD_MIP", [llvm_any_ty], [],
  671. [IntrReadMem, IntrWillReturn], [SDNPMemOperand], 1>,
  672. AMDGPUImageDMaskIntrinsic;
  673. defm int_amdgcn_image_store : AMDGPUImageDimIntrinsicsAll<
  674. "STORE", [], [AMDGPUArg<llvm_anyfloat_ty, "vdata">],
  675. [IntrWriteMem, IntrWillReturn], [SDNPMemOperand]>;
  676. defm int_amdgcn_image_store_mip : AMDGPUImageDimIntrinsicsNoMsaa<
  677. "STORE_MIP", [], [AMDGPUArg<llvm_anyfloat_ty, "vdata">],
  678. [IntrWriteMem, IntrWillReturn], [SDNPMemOperand], 1>;
  679. //////////////////////////////////////////////////////////////////////////
  680. // MSAA intrinsics
  681. //////////////////////////////////////////////////////////////////////////
  682. foreach dim = AMDGPUDims.Msaa in {
  683. def int_amdgcn_image_msaa_load_x # _ # dim.Name:
  684. AMDGPUImageDimIntrinsic<
  685. AMDGPUDimNoSampleProfile<"MSAA_LOAD_X", dim, [llvm_any_ty], []>,
  686. [IntrReadMem], [SDNPMemOperand]>;
  687. }
  688. //////////////////////////////////////////////////////////////////////////
  689. // sample and getlod intrinsics
  690. //////////////////////////////////////////////////////////////////////////
  691. multiclass AMDGPUImageDimSampleDims<string opmod,
  692. AMDGPUSampleVariant sample,
  693. bit NoMem = false> {
  694. foreach dim = AMDGPUDims.NoMsaa in {
  695. def !strconcat(NAME, "_", dim.Name) : AMDGPUImageDimIntrinsic<
  696. AMDGPUDimSampleProfile<opmod, dim, sample>,
  697. !if(NoMem, [IntrNoMem], [IntrReadMem]),
  698. !if(NoMem, [], [SDNPMemOperand])>;
  699. }
  700. }
  701. foreach sample = AMDGPUSampleVariants in {
  702. defm int_amdgcn_image_sample # sample.LowerCaseMod
  703. : AMDGPUImageDimSampleDims<"SAMPLE" # sample.UpperCaseMod, sample>,
  704. AMDGPUImageDMaskIntrinsic;
  705. }
  706. defm int_amdgcn_image_getlod
  707. : AMDGPUImageDimSampleDims<"GET_LOD", AMDGPUSample, 1>,
  708. AMDGPUImageDMaskIntrinsic;
  709. //////////////////////////////////////////////////////////////////////////
  710. // getresinfo intrinsics
  711. //////////////////////////////////////////////////////////////////////////
  712. foreach dim = AMDGPUDims.All in {
  713. def !strconcat("int_amdgcn_image_getresinfo_", dim.Name)
  714. : AMDGPUImageDimIntrinsic<AMDGPUDimGetResInfoProfile<dim>, [IntrNoMem], []>,
  715. AMDGPUImageDMaskIntrinsic;
  716. }
  717. //////////////////////////////////////////////////////////////////////////
  718. // gather4 intrinsics
  719. //////////////////////////////////////////////////////////////////////////
  720. foreach sample = AMDGPUSampleVariantsNoGradients in {
  721. foreach dim = [AMDGPUDim2D, AMDGPUDimCube, AMDGPUDim2DArray] in {
  722. def int_amdgcn_image_gather4 # sample.LowerCaseMod # _ # dim.Name:
  723. AMDGPUImageDimIntrinsic<
  724. AMDGPUDimSampleProfile<"GATHER4" # sample.UpperCaseMod, dim, sample>,
  725. [IntrReadMem], [SDNPMemOperand]>;
  726. }
  727. }
  728. }
  729. //////////////////////////////////////////////////////////////////////////
  730. // atomic intrinsics
  731. //////////////////////////////////////////////////////////////////////////
  732. defset list<AMDGPUImageDimIntrinsic> AMDGPUImageDimAtomicIntrinsics = {
  733. multiclass AMDGPUImageDimAtomicX<string opmod, list<AMDGPUArg> dataargs> {
  734. foreach dim = AMDGPUDims.All in {
  735. def !strconcat(NAME, "_", dim.Name)
  736. : AMDGPUImageDimIntrinsic<
  737. AMDGPUDimAtomicProfile<opmod, dim, dataargs>,
  738. [], [SDNPMemOperand]>;
  739. }
  740. }
  741. multiclass AMDGPUImageDimAtomic<string opmod> {
  742. defm "" : AMDGPUImageDimAtomicX<opmod, [AMDGPUArg<LLVMMatchType<0>, "vdata">]>;
  743. }
  744. defm int_amdgcn_image_atomic_swap : AMDGPUImageDimAtomic<"ATOMIC_SWAP">;
  745. defm int_amdgcn_image_atomic_add : AMDGPUImageDimAtomic<"ATOMIC_ADD">;
  746. defm int_amdgcn_image_atomic_sub : AMDGPUImageDimAtomic<"ATOMIC_SUB">;
  747. defm int_amdgcn_image_atomic_smin : AMDGPUImageDimAtomic<"ATOMIC_SMIN">;
  748. defm int_amdgcn_image_atomic_umin : AMDGPUImageDimAtomic<"ATOMIC_UMIN">;
  749. defm int_amdgcn_image_atomic_smax : AMDGPUImageDimAtomic<"ATOMIC_SMAX">;
  750. defm int_amdgcn_image_atomic_umax : AMDGPUImageDimAtomic<"ATOMIC_UMAX">;
  751. defm int_amdgcn_image_atomic_and : AMDGPUImageDimAtomic<"ATOMIC_AND">;
  752. defm int_amdgcn_image_atomic_or : AMDGPUImageDimAtomic<"ATOMIC_OR">;
  753. defm int_amdgcn_image_atomic_xor : AMDGPUImageDimAtomic<"ATOMIC_XOR">;
  754. defm int_amdgcn_image_atomic_inc : AMDGPUImageDimAtomic<"ATOMIC_INC">;
  755. defm int_amdgcn_image_atomic_dec : AMDGPUImageDimAtomic<"ATOMIC_DEC">;
  756. defm int_amdgcn_image_atomic_cmpswap :
  757. AMDGPUImageDimAtomicX<"ATOMIC_CMPSWAP", [AMDGPUArg<LLVMMatchType<0>, "src">,
  758. AMDGPUArg<LLVMMatchType<0>, "cmp">]>;
  759. }
  760. //////////////////////////////////////////////////////////////////////////
  761. // Buffer intrinsics
  762. //////////////////////////////////////////////////////////////////////////
  763. let TargetPrefix = "amdgcn" in {
  764. defset list<AMDGPURsrcIntrinsic> AMDGPUBufferIntrinsics = {
  765. class AMDGPUBufferLoad<LLVMType data_ty = llvm_any_ty> : Intrinsic <
  766. [data_ty],
  767. [llvm_v4i32_ty, // rsrc(SGPR)
  768. llvm_i32_ty, // vindex(VGPR)
  769. llvm_i32_ty, // offset(SGPR/VGPR/imm)
  770. llvm_i1_ty, // glc(imm)
  771. llvm_i1_ty], // slc(imm)
  772. [IntrReadMem, IntrWillReturn,
  773. ImmArg<ArgIndex<3>>, ImmArg<ArgIndex<4>>], "", [SDNPMemOperand]>,
  774. AMDGPURsrcIntrinsic<0>;
  775. def int_amdgcn_buffer_load_format : AMDGPUBufferLoad<llvm_anyfloat_ty>;
  776. def int_amdgcn_buffer_load : AMDGPUBufferLoad;
  777. def int_amdgcn_s_buffer_load : Intrinsic <
  778. [llvm_any_ty],
  779. [llvm_v4i32_ty, // rsrc(SGPR)
  780. llvm_i32_ty, // byte offset(SGPR/imm)
  781. llvm_i32_ty], // cachepolicy(imm; bit 0 = glc, bit 2 = dlc)
  782. [IntrNoMem, IntrWillReturn, ImmArg<ArgIndex<2>>]>,
  783. AMDGPURsrcIntrinsic<0>;
  784. class AMDGPUBufferStore<LLVMType data_ty = llvm_any_ty> : Intrinsic <
  785. [],
  786. [data_ty, // vdata(VGPR)
  787. llvm_v4i32_ty, // rsrc(SGPR)
  788. llvm_i32_ty, // vindex(VGPR)
  789. llvm_i32_ty, // offset(SGPR/VGPR/imm)
  790. llvm_i1_ty, // glc(imm)
  791. llvm_i1_ty], // slc(imm)
  792. [IntrWriteMem, IntrWillReturn,
  793. ImmArg<ArgIndex<4>>, ImmArg<ArgIndex<5>>], "", [SDNPMemOperand]>,
  794. AMDGPURsrcIntrinsic<1>;
  795. def int_amdgcn_buffer_store_format : AMDGPUBufferStore<llvm_anyfloat_ty>;
  796. def int_amdgcn_buffer_store : AMDGPUBufferStore;
  797. // New buffer intrinsics with separate raw and struct variants. The raw
  798. // variant never has an index. The struct variant always has an index, even if
  799. // it is const 0. A struct intrinsic with constant 0 index is different to the
  800. // corresponding raw intrinsic on gfx9+ because the behavior of bound checking
  801. // and swizzling changes depending on whether idxen is set in the instruction.
  802. // These new instrinsics also keep the offset and soffset arguments separate as
  803. // they behave differently in bounds checking and swizzling.
  804. class AMDGPURawBufferLoad<LLVMType data_ty = llvm_any_ty> : Intrinsic <
  805. [data_ty],
  806. [llvm_v4i32_ty, // rsrc(SGPR)
  807. llvm_i32_ty, // offset(VGPR/imm, included in bounds checking and swizzling)
  808. llvm_i32_ty, // soffset(SGPR/imm, excluded from bounds checking and swizzling)
  809. llvm_i32_ty], // auxiliary data (imm, cachepolicy (bit 0 = glc,
  810. // bit 1 = slc,
  811. // bit 2 = dlc on gfx10+),
  812. // swizzled buffer (bit 3 = swz))
  813. [IntrReadMem, IntrWillReturn, ImmArg<ArgIndex<3>>], "", [SDNPMemOperand]>,
  814. AMDGPURsrcIntrinsic<0>;
  815. def int_amdgcn_raw_buffer_load_format : AMDGPURawBufferLoad<llvm_anyfloat_ty>;
  816. def int_amdgcn_raw_buffer_load : AMDGPURawBufferLoad;
  817. class AMDGPUStructBufferLoad<LLVMType data_ty = llvm_any_ty> : Intrinsic <
  818. [data_ty],
  819. [llvm_v4i32_ty, // rsrc(SGPR)
  820. llvm_i32_ty, // vindex(VGPR)
  821. llvm_i32_ty, // offset(VGPR/imm, included in bounds checking and swizzling)
  822. llvm_i32_ty, // soffset(SGPR/imm, excluded from bounds checking and swizzling)
  823. llvm_i32_ty], // auxiliary data (imm, cachepolicy (bit 0 = glc,
  824. // bit 1 = slc,
  825. // bit 2 = dlc on gfx10+),
  826. // swizzled buffer (bit 3 = swz))
  827. [IntrReadMem, IntrWillReturn, ImmArg<ArgIndex<4>>], "", [SDNPMemOperand]>,
  828. AMDGPURsrcIntrinsic<0>;
  829. def int_amdgcn_struct_buffer_load_format : AMDGPUStructBufferLoad;
  830. def int_amdgcn_struct_buffer_load : AMDGPUStructBufferLoad;
  831. class AMDGPURawBufferStore<LLVMType data_ty = llvm_any_ty> : Intrinsic <
  832. [],
  833. [data_ty, // vdata(VGPR)
  834. llvm_v4i32_ty, // rsrc(SGPR)
  835. llvm_i32_ty, // offset(VGPR/imm, included in bounds checking and swizzling)
  836. llvm_i32_ty, // soffset(SGPR/imm, excluded from bounds checking and swizzling)
  837. llvm_i32_ty], // auxiliary data (imm, cachepolicy (bit 0 = glc,
  838. // bit 1 = slc,
  839. // bit 2 = dlc on gfx10+),
  840. // swizzled buffer (bit 3 = swz))
  841. [IntrWriteMem, IntrWillReturn, ImmArg<ArgIndex<4>>], "", [SDNPMemOperand]>,
  842. AMDGPURsrcIntrinsic<1>;
  843. def int_amdgcn_raw_buffer_store_format : AMDGPURawBufferStore<llvm_anyfloat_ty>;
  844. def int_amdgcn_raw_buffer_store : AMDGPURawBufferStore;
  845. class AMDGPUStructBufferStore<LLVMType data_ty = llvm_any_ty> : Intrinsic <
  846. [],
  847. [data_ty, // vdata(VGPR)
  848. llvm_v4i32_ty, // rsrc(SGPR)
  849. llvm_i32_ty, // vindex(VGPR)
  850. llvm_i32_ty, // offset(VGPR/imm, included in bounds checking and swizzling)
  851. llvm_i32_ty, // soffset(SGPR/imm, excluded from bounds checking and swizzling)
  852. llvm_i32_ty], // auxiliary data (imm, cachepolicy (bit 0 = glc,
  853. // bit 1 = slc,
  854. // bit 2 = dlc on gfx10+),
  855. // swizzled buffer (bit 3 = swz))
  856. [IntrWriteMem, IntrWillReturn, ImmArg<ArgIndex<5>>], "", [SDNPMemOperand]>,
  857. AMDGPURsrcIntrinsic<1>;
  858. def int_amdgcn_struct_buffer_store_format : AMDGPUStructBufferStore;
  859. def int_amdgcn_struct_buffer_store : AMDGPUStructBufferStore;
  860. class AMDGPURawBufferAtomic<LLVMType data_ty = llvm_any_ty, bit NoRtn = false> : Intrinsic <
  861. !if(NoRtn, [], [data_ty]),
  862. [!if(NoRtn, data_ty, LLVMMatchType<0>), // vdata(VGPR)
  863. llvm_v4i32_ty, // rsrc(SGPR)
  864. llvm_i32_ty, // offset(VGPR/imm, included in bounds checking and swizzling)
  865. llvm_i32_ty, // soffset(SGPR/imm, excluded from bounds checking and swizzling)
  866. llvm_i32_ty], // cachepolicy(imm; bit 1 = slc)
  867. [ImmArg<ArgIndex<4>>, IntrWillReturn], "", [SDNPMemOperand]>,
  868. AMDGPURsrcIntrinsic<1, 0>;
  869. def int_amdgcn_raw_buffer_atomic_swap : AMDGPURawBufferAtomic;
  870. def int_amdgcn_raw_buffer_atomic_add : AMDGPURawBufferAtomic;
  871. def int_amdgcn_raw_buffer_atomic_sub : AMDGPURawBufferAtomic;
  872. def int_amdgcn_raw_buffer_atomic_smin : AMDGPURawBufferAtomic;
  873. def int_amdgcn_raw_buffer_atomic_umin : AMDGPURawBufferAtomic;
  874. def int_amdgcn_raw_buffer_atomic_smax : AMDGPURawBufferAtomic;
  875. def int_amdgcn_raw_buffer_atomic_umax : AMDGPURawBufferAtomic;
  876. def int_amdgcn_raw_buffer_atomic_and : AMDGPURawBufferAtomic;
  877. def int_amdgcn_raw_buffer_atomic_or : AMDGPURawBufferAtomic;
  878. def int_amdgcn_raw_buffer_atomic_xor : AMDGPURawBufferAtomic;
  879. def int_amdgcn_raw_buffer_atomic_inc : AMDGPURawBufferAtomic;
  880. def int_amdgcn_raw_buffer_atomic_dec : AMDGPURawBufferAtomic;
  881. def int_amdgcn_raw_buffer_atomic_cmpswap : Intrinsic<
  882. [llvm_anyint_ty],
  883. [LLVMMatchType<0>, // src(VGPR)
  884. LLVMMatchType<0>, // cmp(VGPR)
  885. llvm_v4i32_ty, // rsrc(SGPR)
  886. llvm_i32_ty, // offset(VGPR/imm, included in bounds checking and swizzling)
  887. llvm_i32_ty, // soffset(SGPR/imm, excluded from bounds checking and swizzling)
  888. llvm_i32_ty], // cachepolicy(imm; bit 1 = slc)
  889. [ImmArg<ArgIndex<5>>, IntrWillReturn], "", [SDNPMemOperand]>,
  890. AMDGPURsrcIntrinsic<2, 0>;
  891. // gfx908 intrinsic
  892. def int_amdgcn_raw_buffer_atomic_fadd : AMDGPURawBufferAtomic<llvm_anyfloat_ty>;
  893. // gfx90a intrinsics
  894. def int_amdgcn_raw_buffer_atomic_fmin : AMDGPURawBufferAtomic<llvm_anyfloat_ty>;
  895. def int_amdgcn_raw_buffer_atomic_fmax : AMDGPURawBufferAtomic<llvm_anyfloat_ty>;
  896. class AMDGPUStructBufferAtomic<LLVMType data_ty = llvm_any_ty, bit NoRtn = false> : Intrinsic <
  897. !if(NoRtn, [], [data_ty]),
  898. [!if(NoRtn, data_ty, LLVMMatchType<0>), // vdata(VGPR)
  899. llvm_v4i32_ty, // rsrc(SGPR)
  900. llvm_i32_ty, // vindex(VGPR)
  901. llvm_i32_ty, // offset(VGPR/imm, included in bounds checking and swizzling)
  902. llvm_i32_ty, // soffset(SGPR/imm, excluded from bounds checking and swizzling)
  903. llvm_i32_ty], // cachepolicy(imm; bit 1 = slc)
  904. [ImmArg<ArgIndex<5>>, IntrWillReturn], "", [SDNPMemOperand]>,
  905. AMDGPURsrcIntrinsic<1, 0>;
  906. def int_amdgcn_struct_buffer_atomic_swap : AMDGPUStructBufferAtomic;
  907. def int_amdgcn_struct_buffer_atomic_add : AMDGPUStructBufferAtomic;
  908. def int_amdgcn_struct_buffer_atomic_sub : AMDGPUStructBufferAtomic;
  909. def int_amdgcn_struct_buffer_atomic_smin : AMDGPUStructBufferAtomic;
  910. def int_amdgcn_struct_buffer_atomic_umin : AMDGPUStructBufferAtomic;
  911. def int_amdgcn_struct_buffer_atomic_smax : AMDGPUStructBufferAtomic;
  912. def int_amdgcn_struct_buffer_atomic_umax : AMDGPUStructBufferAtomic;
  913. def int_amdgcn_struct_buffer_atomic_and : AMDGPUStructBufferAtomic;
  914. def int_amdgcn_struct_buffer_atomic_or : AMDGPUStructBufferAtomic;
  915. def int_amdgcn_struct_buffer_atomic_xor : AMDGPUStructBufferAtomic;
  916. def int_amdgcn_struct_buffer_atomic_inc : AMDGPUStructBufferAtomic;
  917. def int_amdgcn_struct_buffer_atomic_dec : AMDGPUStructBufferAtomic;
  918. def int_amdgcn_struct_buffer_atomic_cmpswap : Intrinsic<
  919. [llvm_anyint_ty],
  920. [LLVMMatchType<0>, // src(VGPR)
  921. LLVMMatchType<0>, // cmp(VGPR)
  922. llvm_v4i32_ty, // rsrc(SGPR)
  923. llvm_i32_ty, // vindex(VGPR)
  924. llvm_i32_ty, // offset(VGPR/imm, included in bounds checking and swizzling)
  925. llvm_i32_ty, // soffset(SGPR/imm, excluded from bounds checking and swizzling)
  926. llvm_i32_ty], // cachepolicy(imm; bit 1 = slc)
  927. [ImmArg<ArgIndex<6>>, IntrWillReturn], "", [SDNPMemOperand]>,
  928. AMDGPURsrcIntrinsic<2, 0>;
  929. // gfx908 intrinsic
  930. def int_amdgcn_struct_buffer_atomic_fadd : AMDGPUStructBufferAtomic<llvm_anyfloat_ty>;
  931. // gfx90a intrinsics
  932. def int_amdgcn_struct_buffer_atomic_fmin : AMDGPUStructBufferAtomic<llvm_anyfloat_ty>;
  933. def int_amdgcn_struct_buffer_atomic_fmax : AMDGPUStructBufferAtomic<llvm_anyfloat_ty>;
  934. // Obsolescent tbuffer intrinsics.
  935. def int_amdgcn_tbuffer_load : Intrinsic <
  936. [llvm_any_ty], // overloaded for types f32/i32, v2f32/v2i32, v4f32/v4i32
  937. [llvm_v4i32_ty, // rsrc(SGPR)
  938. llvm_i32_ty, // vindex(VGPR)
  939. llvm_i32_ty, // voffset(VGPR)
  940. llvm_i32_ty, // soffset(SGPR)
  941. llvm_i32_ty, // offset(imm)
  942. llvm_i32_ty, // dfmt(imm)
  943. llvm_i32_ty, // nfmt(imm)
  944. llvm_i1_ty, // glc(imm)
  945. llvm_i1_ty], // slc(imm)
  946. [IntrReadMem, IntrWillReturn,
  947. ImmArg<ArgIndex<4>>, ImmArg<ArgIndex<5>>, ImmArg<ArgIndex<6>>,
  948. ImmArg<ArgIndex<7>>, ImmArg<ArgIndex<8>>], "", [SDNPMemOperand]>,
  949. AMDGPURsrcIntrinsic<0>;
  950. def int_amdgcn_tbuffer_store : Intrinsic <
  951. [],
  952. [llvm_any_ty, // vdata(VGPR), overloaded for types f32/i32, v2f32/v2i32, v4f32/v4i32
  953. llvm_v4i32_ty, // rsrc(SGPR)
  954. llvm_i32_ty, // vindex(VGPR)
  955. llvm_i32_ty, // voffset(VGPR)
  956. llvm_i32_ty, // soffset(SGPR)
  957. llvm_i32_ty, // offset(imm)
  958. llvm_i32_ty, // dfmt(imm)
  959. llvm_i32_ty, // nfmt(imm)
  960. llvm_i1_ty, // glc(imm)
  961. llvm_i1_ty], // slc(imm)
  962. [IntrWriteMem, IntrWillReturn, ImmArg<ArgIndex<5>>,
  963. ImmArg<ArgIndex<6>>, ImmArg<ArgIndex<7>>,
  964. ImmArg<ArgIndex<8>>, ImmArg<ArgIndex<9>>], "", [SDNPMemOperand]>,
  965. AMDGPURsrcIntrinsic<1>;
  966. // New tbuffer intrinsics, with:
  967. // - raw and struct variants
  968. // - joint format field
  969. // - joint cachepolicy field
  970. def int_amdgcn_raw_tbuffer_load : Intrinsic <
  971. [llvm_any_ty], // overloaded for types f32/i32, v2f32/v2i32, v4f32/v4i32
  972. [llvm_v4i32_ty, // rsrc(SGPR)
  973. llvm_i32_ty, // offset(VGPR/imm, included in bounds checking and swizzling)
  974. llvm_i32_ty, // soffset(SGPR/imm, excluded from bounds checking and swizzling)
  975. llvm_i32_ty, // format(imm; bits 3..0 = dfmt, bits 6..4 = nfmt)
  976. llvm_i32_ty], // auxiliary data (imm, cachepolicy (bit 0 = glc,
  977. // bit 1 = slc,
  978. // bit 2 = dlc on gfx10+),
  979. // swizzled buffer (bit 3 = swz))
  980. [IntrReadMem, IntrWillReturn,
  981. ImmArg<ArgIndex<3>>, ImmArg<ArgIndex<4>>], "", [SDNPMemOperand]>,
  982. AMDGPURsrcIntrinsic<0>;
  983. def int_amdgcn_raw_tbuffer_store : Intrinsic <
  984. [],
  985. [llvm_any_ty, // vdata(VGPR), overloaded for types f32/i32, v2f32/v2i32, v4f32/v4i32
  986. llvm_v4i32_ty, // rsrc(SGPR)
  987. llvm_i32_ty, // offset(VGPR/imm, included in bounds checking and swizzling)
  988. llvm_i32_ty, // soffset(SGPR/imm, excluded from bounds checking and swizzling)
  989. llvm_i32_ty, // format(imm; bits 3..0 = dfmt, bits 6..4 = nfmt)
  990. llvm_i32_ty], // auxiliary data (imm, cachepolicy (bit 0 = glc,
  991. // bit 1 = slc,
  992. // bit 2 = dlc on gfx10+),
  993. // swizzled buffer (bit 3 = swz))
  994. [IntrWriteMem, IntrWillReturn,
  995. ImmArg<ArgIndex<4>>, ImmArg<ArgIndex<5>>], "", [SDNPMemOperand]>,
  996. AMDGPURsrcIntrinsic<1>;
  997. def int_amdgcn_struct_tbuffer_load : Intrinsic <
  998. [llvm_any_ty], // overloaded for types f32/i32, v2f32/v2i32, v4f32/v4i32
  999. [llvm_v4i32_ty, // rsrc(SGPR)
  1000. llvm_i32_ty, // vindex(VGPR)
  1001. llvm_i32_ty, // offset(VGPR/imm, included in bounds checking and swizzling)
  1002. llvm_i32_ty, // soffset(SGPR/imm, excluded from bounds checking and swizzling)
  1003. llvm_i32_ty, // format(imm; bits 3..0 = dfmt, bits 6..4 = nfmt)
  1004. llvm_i32_ty], // auxiliary data (imm, cachepolicy (bit 0 = glc,
  1005. // bit 1 = slc,
  1006. // bit 2 = dlc on gfx10+),
  1007. // swizzled buffer (bit 3 = swz))
  1008. [IntrReadMem, IntrWillReturn,
  1009. ImmArg<ArgIndex<4>>, ImmArg<ArgIndex<5>>], "", [SDNPMemOperand]>,
  1010. AMDGPURsrcIntrinsic<0>;
  1011. def int_amdgcn_struct_tbuffer_store : Intrinsic <
  1012. [],
  1013. [llvm_any_ty, // vdata(VGPR), overloaded for types f32/i32, v2f32/v2i32, v4f32/v4i32
  1014. llvm_v4i32_ty, // rsrc(SGPR)
  1015. llvm_i32_ty, // vindex(VGPR)
  1016. llvm_i32_ty, // offset(VGPR/imm, included in bounds checking and swizzling)
  1017. llvm_i32_ty, // soffset(SGPR/imm, excluded from bounds checking and swizzling)
  1018. llvm_i32_ty, // format(imm; bits 3..0 = dfmt, bits 6..4 = nfmt)
  1019. llvm_i32_ty], // auxiliary data (imm, cachepolicy (bit 0 = glc,
  1020. // bit 1 = slc,
  1021. // bit 2 = dlc on gfx10+),
  1022. // swizzled buffer (bit 3 = swz))
  1023. [IntrWriteMem, IntrWillReturn,
  1024. ImmArg<ArgIndex<5>>, ImmArg<ArgIndex<6>>], "", [SDNPMemOperand]>,
  1025. AMDGPURsrcIntrinsic<1>;
  1026. class AMDGPUBufferAtomic : Intrinsic <
  1027. [llvm_anyint_ty],
  1028. [LLVMMatchType<0>, // vdata(VGPR)
  1029. llvm_v4i32_ty, // rsrc(SGPR)
  1030. llvm_i32_ty, // vindex(VGPR)
  1031. llvm_i32_ty, // offset(SGPR/VGPR/imm)
  1032. llvm_i1_ty], // slc(imm)
  1033. [ImmArg<ArgIndex<4>>, IntrWillReturn], "", [SDNPMemOperand]>,
  1034. AMDGPURsrcIntrinsic<1, 0>;
  1035. def int_amdgcn_buffer_atomic_swap : AMDGPUBufferAtomic;
  1036. def int_amdgcn_buffer_atomic_add : AMDGPUBufferAtomic;
  1037. def int_amdgcn_buffer_atomic_sub : AMDGPUBufferAtomic;
  1038. def int_amdgcn_buffer_atomic_smin : AMDGPUBufferAtomic;
  1039. def int_amdgcn_buffer_atomic_umin : AMDGPUBufferAtomic;
  1040. def int_amdgcn_buffer_atomic_smax : AMDGPUBufferAtomic;
  1041. def int_amdgcn_buffer_atomic_umax : AMDGPUBufferAtomic;
  1042. def int_amdgcn_buffer_atomic_and : AMDGPUBufferAtomic;
  1043. def int_amdgcn_buffer_atomic_or : AMDGPUBufferAtomic;
  1044. def int_amdgcn_buffer_atomic_xor : AMDGPUBufferAtomic;
  1045. def int_amdgcn_buffer_atomic_cmpswap : Intrinsic<
  1046. [llvm_i32_ty],
  1047. [llvm_i32_ty, // src(VGPR)
  1048. llvm_i32_ty, // cmp(VGPR)
  1049. llvm_v4i32_ty, // rsrc(SGPR)
  1050. llvm_i32_ty, // vindex(VGPR)
  1051. llvm_i32_ty, // offset(SGPR/VGPR/imm)
  1052. llvm_i1_ty], // slc(imm)
  1053. [ImmArg<ArgIndex<5>>, IntrWillReturn], "", [SDNPMemOperand]>,
  1054. AMDGPURsrcIntrinsic<2, 0>;
  1055. def int_amdgcn_buffer_atomic_csub : AMDGPUBufferAtomic;
  1056. class AMDGPUBufferAtomicFP : Intrinsic <
  1057. [llvm_anyfloat_ty],
  1058. [LLVMMatchType<0>, // vdata(VGPR)
  1059. llvm_v4i32_ty, // rsrc(SGPR)
  1060. llvm_i32_ty, // vindex(VGPR)
  1061. llvm_i32_ty, // offset(SGPR/VGPR/imm)
  1062. llvm_i1_ty], // slc(imm)
  1063. [ImmArg<ArgIndex<4>>, IntrWillReturn], "", [SDNPMemOperand]>,
  1064. AMDGPURsrcIntrinsic<1, 0>;
  1065. // Legacy form of the intrinsic. raw and struct forms should be preferred.
  1066. def int_amdgcn_buffer_atomic_fadd : AMDGPUBufferAtomicFP;
  1067. } // defset AMDGPUBufferIntrinsics
  1068. // Uses that do not set the done bit should set IntrWriteMem on the
  1069. // call site.
  1070. def int_amdgcn_exp : Intrinsic <[], [
  1071. llvm_i32_ty, // tgt,
  1072. llvm_i32_ty, // en
  1073. llvm_any_ty, // src0 (f32 or i32)
  1074. LLVMMatchType<0>, // src1
  1075. LLVMMatchType<0>, // src2
  1076. LLVMMatchType<0>, // src3
  1077. llvm_i1_ty, // done
  1078. llvm_i1_ty // vm
  1079. ],
  1080. [ImmArg<ArgIndex<0>>, ImmArg<ArgIndex<1>>, ImmArg<ArgIndex<6>>,
  1081. ImmArg<ArgIndex<7>>, IntrWriteMem, IntrInaccessibleMemOnly,
  1082. IntrWillReturn]
  1083. >;
  1084. // exp with compr bit set.
  1085. def int_amdgcn_exp_compr : Intrinsic <[], [
  1086. llvm_i32_ty, // tgt,
  1087. llvm_i32_ty, // en
  1088. llvm_anyvector_ty, // src0 (v2f16 or v2i16)
  1089. LLVMMatchType<0>, // src1
  1090. llvm_i1_ty, // done
  1091. llvm_i1_ty], // vm
  1092. [ImmArg<ArgIndex<0>>, ImmArg<ArgIndex<1>>, ImmArg<ArgIndex<4>>,
  1093. ImmArg<ArgIndex<5>>, IntrWriteMem, IntrInaccessibleMemOnly,
  1094. IntrWillReturn]
  1095. >;
  1096. def int_amdgcn_buffer_wbinvl1_sc :
  1097. GCCBuiltin<"__builtin_amdgcn_buffer_wbinvl1_sc">,
  1098. Intrinsic<[], [], [IntrNoMem, IntrHasSideEffects, IntrWillReturn]>;
  1099. def int_amdgcn_buffer_wbinvl1 :
  1100. GCCBuiltin<"__builtin_amdgcn_buffer_wbinvl1">,
  1101. Intrinsic<[], [], [IntrNoMem, IntrHasSideEffects, IntrWillReturn]>;
  1102. def int_amdgcn_s_dcache_inv :
  1103. GCCBuiltin<"__builtin_amdgcn_s_dcache_inv">,
  1104. Intrinsic<[], [], [IntrNoMem, IntrHasSideEffects, IntrWillReturn]>;
  1105. def int_amdgcn_s_memtime :
  1106. GCCBuiltin<"__builtin_amdgcn_s_memtime">,
  1107. Intrinsic<[llvm_i64_ty], [], [IntrWillReturn]>;
  1108. def int_amdgcn_s_sleep :
  1109. GCCBuiltin<"__builtin_amdgcn_s_sleep">,
  1110. Intrinsic<[], [llvm_i32_ty], [ImmArg<ArgIndex<0>>, IntrNoMem,
  1111. IntrHasSideEffects, IntrWillReturn]> {
  1112. }
  1113. def int_amdgcn_s_incperflevel :
  1114. GCCBuiltin<"__builtin_amdgcn_s_incperflevel">,
  1115. Intrinsic<[], [llvm_i32_ty], [ImmArg<ArgIndex<0>>, IntrNoMem,
  1116. IntrHasSideEffects, IntrWillReturn]> {
  1117. }
  1118. def int_amdgcn_s_decperflevel :
  1119. GCCBuiltin<"__builtin_amdgcn_s_decperflevel">,
  1120. Intrinsic<[], [llvm_i32_ty], [ImmArg<ArgIndex<0>>, IntrNoMem,
  1121. IntrHasSideEffects, IntrWillReturn]> {
  1122. }
  1123. def int_amdgcn_s_sethalt :
  1124. Intrinsic<[], [llvm_i32_ty], [ImmArg<ArgIndex<0>>, IntrNoMem,
  1125. IntrHasSideEffects, IntrWillReturn]>;
  1126. def int_amdgcn_s_getreg :
  1127. GCCBuiltin<"__builtin_amdgcn_s_getreg">,
  1128. Intrinsic<[llvm_i32_ty], [llvm_i32_ty],
  1129. [IntrInaccessibleMemOnly, IntrReadMem, IntrSpeculatable,
  1130. IntrWillReturn, ImmArg<ArgIndex<0>>]
  1131. >;
  1132. // Note this can be used to set FP environment properties that are
  1133. // unsafe to change in non-strictfp functions. The register properties
  1134. // available (and value required to access them) may differ per
  1135. // subtarget. llvm.amdgcn.s.setreg(hwmode, value)
  1136. def int_amdgcn_s_setreg :
  1137. GCCBuiltin<"__builtin_amdgcn_s_setreg">,
  1138. Intrinsic<[], [llvm_i32_ty, llvm_i32_ty],
  1139. [IntrNoMem, IntrHasSideEffects, IntrWillReturn, ImmArg<ArgIndex<0>>]
  1140. >;
  1141. // int_amdgcn_s_getpc is provided to allow a specific style of position
  1142. // independent code to determine the high part of its address when it is
  1143. // known (through convention) that the code and any data of interest does
  1144. // not cross a 4Gb address boundary. Use for any other purpose may not
  1145. // produce the desired results as optimizations may cause code movement,
  1146. // especially as we explicitly use IntrNoMem to allow optimizations.
  1147. def int_amdgcn_s_getpc :
  1148. GCCBuiltin<"__builtin_amdgcn_s_getpc">,
  1149. Intrinsic<[llvm_i64_ty], [], [IntrNoMem, IntrSpeculatable,
  1150. IntrWillReturn]>;
  1151. // __builtin_amdgcn_interp_mov <param>, <attr_chan>, <attr>, <m0>
  1152. // param values: 0 = P10, 1 = P20, 2 = P0
  1153. def int_amdgcn_interp_mov :
  1154. GCCBuiltin<"__builtin_amdgcn_interp_mov">,
  1155. Intrinsic<[llvm_float_ty],
  1156. [llvm_i32_ty, llvm_i32_ty, llvm_i32_ty, llvm_i32_ty],
  1157. [IntrNoMem, IntrSpeculatable, IntrWillReturn,
  1158. ImmArg<ArgIndex<0>>, ImmArg<ArgIndex<1>>, ImmArg<ArgIndex<2>>]>;
  1159. // __builtin_amdgcn_interp_p1 <i>, <attr_chan>, <attr>, <m0>
  1160. // This intrinsic reads from lds, but the memory values are constant,
  1161. // so it behaves like IntrNoMem.
  1162. def int_amdgcn_interp_p1 :
  1163. GCCBuiltin<"__builtin_amdgcn_interp_p1">,
  1164. Intrinsic<[llvm_float_ty],
  1165. [llvm_float_ty, llvm_i32_ty, llvm_i32_ty, llvm_i32_ty],
  1166. [IntrNoMem, IntrSpeculatable, IntrWillReturn,
  1167. ImmArg<ArgIndex<1>>, ImmArg<ArgIndex<2>>]>;
  1168. // __builtin_amdgcn_interp_p2 <p1>, <j>, <attr_chan>, <attr>, <m0>
  1169. def int_amdgcn_interp_p2 :
  1170. GCCBuiltin<"__builtin_amdgcn_interp_p2">,
  1171. Intrinsic<[llvm_float_ty],
  1172. [llvm_float_ty, llvm_float_ty, llvm_i32_ty, llvm_i32_ty, llvm_i32_ty],
  1173. [IntrNoMem, IntrSpeculatable, IntrWillReturn,
  1174. ImmArg<ArgIndex<2>>, ImmArg<ArgIndex<3>>]>;
  1175. // See int_amdgcn_v_interp_p1 for why this is IntrNoMem.
  1176. // __builtin_amdgcn_interp_p1_f16 <i>, <attr_chan>, <attr>, <high>, <m0>
  1177. // high selects whether high or low 16-bits are loaded from LDS
  1178. def int_amdgcn_interp_p1_f16 :
  1179. GCCBuiltin<"__builtin_amdgcn_interp_p1_f16">,
  1180. Intrinsic<[llvm_float_ty],
  1181. [llvm_float_ty, llvm_i32_ty, llvm_i32_ty, llvm_i1_ty, llvm_i32_ty],
  1182. [IntrNoMem, IntrSpeculatable, IntrWillReturn,
  1183. ImmArg<ArgIndex<1>>, ImmArg<ArgIndex<2>>, ImmArg<ArgIndex<3>>]>;
  1184. // __builtin_amdgcn_interp_p2_f16 <p1>, <j>, <attr_chan>, <attr>, <high>, <m0>
  1185. // high selects whether high or low 16-bits are loaded from LDS
  1186. def int_amdgcn_interp_p2_f16 :
  1187. GCCBuiltin<"__builtin_amdgcn_interp_p2_f16">,
  1188. Intrinsic<[llvm_half_ty],
  1189. [llvm_float_ty, llvm_float_ty, llvm_i32_ty, llvm_i32_ty, llvm_i1_ty, llvm_i32_ty],
  1190. [IntrNoMem, IntrSpeculatable, IntrWillReturn,
  1191. ImmArg<ArgIndex<2>>, ImmArg<ArgIndex<3>>, ImmArg<ArgIndex<4>>]>;
  1192. // Deprecated: use llvm.amdgcn.live.mask instead.
  1193. def int_amdgcn_ps_live : Intrinsic <
  1194. [llvm_i1_ty],
  1195. [],
  1196. [IntrNoMem, IntrWillReturn]>;
  1197. // Query currently live lanes.
  1198. // Returns true if lane is live (and not a helper lane).
  1199. def int_amdgcn_live_mask : Intrinsic <[llvm_i1_ty],
  1200. [], [IntrReadMem, IntrInaccessibleMemOnly, IntrWillReturn]
  1201. >;
  1202. def int_amdgcn_mbcnt_lo :
  1203. GCCBuiltin<"__builtin_amdgcn_mbcnt_lo">,
  1204. Intrinsic<[llvm_i32_ty], [llvm_i32_ty, llvm_i32_ty],
  1205. [IntrNoMem, IntrWillReturn]>;
  1206. def int_amdgcn_mbcnt_hi :
  1207. GCCBuiltin<"__builtin_amdgcn_mbcnt_hi">,
  1208. Intrinsic<[llvm_i32_ty], [llvm_i32_ty, llvm_i32_ty],
  1209. [IntrNoMem, IntrWillReturn]>;
  1210. // llvm.amdgcn.ds.swizzle src offset
  1211. def int_amdgcn_ds_swizzle :
  1212. GCCBuiltin<"__builtin_amdgcn_ds_swizzle">,
  1213. Intrinsic<[llvm_i32_ty], [llvm_i32_ty, llvm_i32_ty],
  1214. [IntrNoMem, IntrConvergent, IntrWillReturn,
  1215. ImmArg<ArgIndex<1>>]>;
  1216. def int_amdgcn_ubfe : Intrinsic<[llvm_anyint_ty],
  1217. [LLVMMatchType<0>, llvm_i32_ty, llvm_i32_ty],
  1218. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  1219. >;
  1220. def int_amdgcn_sbfe : Intrinsic<[llvm_anyint_ty],
  1221. [LLVMMatchType<0>, llvm_i32_ty, llvm_i32_ty],
  1222. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  1223. >;
  1224. def int_amdgcn_lerp :
  1225. GCCBuiltin<"__builtin_amdgcn_lerp">,
  1226. Intrinsic<[llvm_i32_ty], [llvm_i32_ty, llvm_i32_ty, llvm_i32_ty],
  1227. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  1228. >;
  1229. def int_amdgcn_sad_u8 :
  1230. GCCBuiltin<"__builtin_amdgcn_sad_u8">,
  1231. Intrinsic<[llvm_i32_ty], [llvm_i32_ty, llvm_i32_ty, llvm_i32_ty],
  1232. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  1233. >;
  1234. def int_amdgcn_msad_u8 :
  1235. GCCBuiltin<"__builtin_amdgcn_msad_u8">,
  1236. Intrinsic<[llvm_i32_ty], [llvm_i32_ty, llvm_i32_ty, llvm_i32_ty],
  1237. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  1238. >;
  1239. def int_amdgcn_sad_hi_u8 :
  1240. GCCBuiltin<"__builtin_amdgcn_sad_hi_u8">,
  1241. Intrinsic<[llvm_i32_ty], [llvm_i32_ty, llvm_i32_ty, llvm_i32_ty],
  1242. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  1243. >;
  1244. def int_amdgcn_sad_u16 :
  1245. GCCBuiltin<"__builtin_amdgcn_sad_u16">,
  1246. Intrinsic<[llvm_i32_ty], [llvm_i32_ty, llvm_i32_ty, llvm_i32_ty],
  1247. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  1248. >;
  1249. def int_amdgcn_qsad_pk_u16_u8 :
  1250. GCCBuiltin<"__builtin_amdgcn_qsad_pk_u16_u8">,
  1251. Intrinsic<[llvm_i64_ty], [llvm_i64_ty, llvm_i32_ty, llvm_i64_ty],
  1252. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  1253. >;
  1254. def int_amdgcn_mqsad_pk_u16_u8 :
  1255. GCCBuiltin<"__builtin_amdgcn_mqsad_pk_u16_u8">,
  1256. Intrinsic<[llvm_i64_ty], [llvm_i64_ty, llvm_i32_ty, llvm_i64_ty],
  1257. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  1258. >;
  1259. def int_amdgcn_mqsad_u32_u8 :
  1260. GCCBuiltin<"__builtin_amdgcn_mqsad_u32_u8">,
  1261. Intrinsic<[llvm_v4i32_ty], [llvm_i64_ty, llvm_i32_ty, llvm_v4i32_ty],
  1262. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  1263. >;
  1264. def int_amdgcn_cvt_pk_u8_f32 :
  1265. GCCBuiltin<"__builtin_amdgcn_cvt_pk_u8_f32">,
  1266. Intrinsic<[llvm_i32_ty], [llvm_float_ty, llvm_i32_ty, llvm_i32_ty],
  1267. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  1268. >;
  1269. def int_amdgcn_icmp :
  1270. Intrinsic<[llvm_anyint_ty], [llvm_anyint_ty, LLVMMatchType<1>, llvm_i32_ty],
  1271. [IntrNoMem, IntrConvergent, IntrWillReturn,
  1272. ImmArg<ArgIndex<2>>]>;
  1273. def int_amdgcn_fcmp :
  1274. Intrinsic<[llvm_anyint_ty], [llvm_anyfloat_ty, LLVMMatchType<1>, llvm_i32_ty],
  1275. [IntrNoMem, IntrConvergent, IntrWillReturn,
  1276. ImmArg<ArgIndex<2>>]>;
  1277. def int_amdgcn_ballot :
  1278. Intrinsic<[llvm_anyint_ty], [llvm_i1_ty],
  1279. [IntrNoMem, IntrConvergent, IntrWillReturn]>;
  1280. def int_amdgcn_readfirstlane :
  1281. GCCBuiltin<"__builtin_amdgcn_readfirstlane">,
  1282. Intrinsic<[llvm_i32_ty], [llvm_i32_ty],
  1283. [IntrNoMem, IntrConvergent, IntrWillReturn]>;
  1284. // The lane argument must be uniform across the currently active threads of the
  1285. // current wave. Otherwise, the result is undefined.
  1286. def int_amdgcn_readlane :
  1287. GCCBuiltin<"__builtin_amdgcn_readlane">,
  1288. Intrinsic<[llvm_i32_ty], [llvm_i32_ty, llvm_i32_ty],
  1289. [IntrNoMem, IntrConvergent, IntrWillReturn]>;
  1290. // The value to write and lane select arguments must be uniform across the
  1291. // currently active threads of the current wave. Otherwise, the result is
  1292. // undefined.
  1293. def int_amdgcn_writelane :
  1294. GCCBuiltin<"__builtin_amdgcn_writelane">,
  1295. Intrinsic<[llvm_i32_ty], [
  1296. llvm_i32_ty, // uniform value to write: returned by the selected lane
  1297. llvm_i32_ty, // uniform lane select
  1298. llvm_i32_ty // returned by all lanes other than the selected one
  1299. ],
  1300. [IntrNoMem, IntrConvergent, IntrWillReturn]
  1301. >;
  1302. // FIXME: Deprecated. This is equivalent to llvm.fshr
  1303. def int_amdgcn_alignbit : Intrinsic<[llvm_i32_ty],
  1304. [llvm_i32_ty, llvm_i32_ty, llvm_i32_ty],
  1305. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  1306. >;
  1307. def int_amdgcn_alignbyte : GCCBuiltin<"__builtin_amdgcn_alignbyte">,
  1308. Intrinsic<[llvm_i32_ty], [llvm_i32_ty, llvm_i32_ty, llvm_i32_ty],
  1309. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  1310. >;
  1311. def int_amdgcn_mul_i24 : Intrinsic<[llvm_i32_ty],
  1312. [llvm_i32_ty, llvm_i32_ty],
  1313. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  1314. >;
  1315. def int_amdgcn_mul_u24 : Intrinsic<[llvm_i32_ty],
  1316. [llvm_i32_ty, llvm_i32_ty],
  1317. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  1318. >;
  1319. // llvm.amdgcn.ds.gws.init(i32 bar_val, i32 resource_id)
  1320. //
  1321. // bar_val is the total number of waves that will wait on this
  1322. // barrier, minus 1.
  1323. def int_amdgcn_ds_gws_init :
  1324. GCCBuiltin<"__builtin_amdgcn_ds_gws_init">,
  1325. Intrinsic<[],
  1326. [llvm_i32_ty, llvm_i32_ty],
  1327. [IntrConvergent, IntrWriteMem,
  1328. IntrInaccessibleMemOnly, IntrWillReturn], "",
  1329. [SDNPMemOperand]
  1330. >;
  1331. // llvm.amdgcn.ds.gws.barrier(i32 vsrc0, i32 resource_id)
  1332. // bar_val is the total number of waves that will wait on this
  1333. // barrier, minus 1.
  1334. def int_amdgcn_ds_gws_barrier :
  1335. GCCBuiltin<"__builtin_amdgcn_ds_gws_barrier">,
  1336. Intrinsic<[],
  1337. [llvm_i32_ty, llvm_i32_ty],
  1338. [IntrConvergent, IntrInaccessibleMemOnly, IntrWillReturn], "",
  1339. [SDNPMemOperand]
  1340. >;
  1341. // llvm.amdgcn.ds.gws.sema.v(i32 resource_id)
  1342. def int_amdgcn_ds_gws_sema_v :
  1343. GCCBuiltin<"__builtin_amdgcn_ds_gws_sema_v">,
  1344. Intrinsic<[],
  1345. [llvm_i32_ty],
  1346. [IntrConvergent, IntrInaccessibleMemOnly, IntrWillReturn], "",
  1347. [SDNPMemOperand]
  1348. >;
  1349. // llvm.amdgcn.ds.gws.sema.br(i32 vsrc, i32 resource_id)
  1350. def int_amdgcn_ds_gws_sema_br :
  1351. GCCBuiltin<"__builtin_amdgcn_ds_gws_sema_br">,
  1352. Intrinsic<[],
  1353. [llvm_i32_ty, llvm_i32_ty],
  1354. [IntrConvergent, IntrInaccessibleMemOnly, IntrWillReturn], "",
  1355. [SDNPMemOperand]
  1356. >;
  1357. // llvm.amdgcn.ds.gws.sema.p(i32 resource_id)
  1358. def int_amdgcn_ds_gws_sema_p :
  1359. GCCBuiltin<"__builtin_amdgcn_ds_gws_sema_p">,
  1360. Intrinsic<[],
  1361. [llvm_i32_ty],
  1362. [IntrConvergent, IntrInaccessibleMemOnly, IntrWillReturn], "",
  1363. [SDNPMemOperand]
  1364. >;
  1365. // llvm.amdgcn.ds.gws.sema.release.all(i32 resource_id)
  1366. def int_amdgcn_ds_gws_sema_release_all :
  1367. GCCBuiltin<"__builtin_amdgcn_ds_gws_sema_release_all">,
  1368. Intrinsic<[],
  1369. [llvm_i32_ty],
  1370. [IntrConvergent, IntrInaccessibleMemOnly, IntrWillReturn], "",
  1371. [SDNPMemOperand]
  1372. >;
  1373. // Copies the source value to the destination value, with the guarantee that
  1374. // the source value is computed as if the entire program were executed in WQM.
  1375. def int_amdgcn_wqm : Intrinsic<[llvm_any_ty],
  1376. [LLVMMatchType<0>], [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  1377. >;
  1378. // Copies the source value to the destination value, such that the source
  1379. // is computed as if the entire program were executed in WQM if any other
  1380. // program code executes in WQM.
  1381. def int_amdgcn_softwqm : Intrinsic<[llvm_any_ty],
  1382. [LLVMMatchType<0>], [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  1383. >;
  1384. // Return true if at least one thread within the pixel quad passes true into
  1385. // the function.
  1386. def int_amdgcn_wqm_vote : Intrinsic<[llvm_i1_ty],
  1387. [llvm_i1_ty], [IntrNoMem, IntrConvergent, IntrWillReturn]
  1388. >;
  1389. // If false, set EXEC=0 for the current thread until the end of program.
  1390. // FIXME: Should this be IntrNoMem, IntrHasSideEffects, or IntrWillReturn?
  1391. def int_amdgcn_kill : Intrinsic<[], [llvm_i1_ty], []>;
  1392. def int_amdgcn_endpgm : GCCBuiltin<"__builtin_amdgcn_endpgm">,
  1393. Intrinsic<[], [], [IntrNoReturn, IntrCold, IntrNoMem, IntrHasSideEffects]
  1394. >;
  1395. // If false, mark all active lanes as helper lanes until the end of program.
  1396. def int_amdgcn_wqm_demote : Intrinsic<[],
  1397. [llvm_i1_ty], [IntrWriteMem, IntrInaccessibleMemOnly]
  1398. >;
  1399. // Copies the active channels of the source value to the destination value,
  1400. // with the guarantee that the source value is computed as if the entire
  1401. // program were executed in Whole Wavefront Mode, i.e. with all channels
  1402. // enabled, with a few exceptions: - Phi nodes which require WWM return an
  1403. // undefined value.
  1404. def int_amdgcn_strict_wwm : Intrinsic<[llvm_any_ty],
  1405. [LLVMMatchType<0>], [IntrNoMem, IntrSpeculatable,
  1406. IntrConvergent, IntrWillReturn]
  1407. >;
  1408. // Deprecated. Use int_amdgcn_strict_wwm instead.
  1409. def int_amdgcn_wwm : Intrinsic<[llvm_any_ty],
  1410. [LLVMMatchType<0>], [IntrNoMem, IntrSpeculatable,
  1411. IntrConvergent, IntrWillReturn]
  1412. >;
  1413. def int_amdgcn_strict_wqm : Intrinsic<[llvm_any_ty],
  1414. [LLVMMatchType<0>], [IntrNoMem, IntrSpeculatable,
  1415. IntrConvergent, IntrWillReturn]
  1416. >;
  1417. // Given a value, copies it while setting all the inactive lanes to a given
  1418. // value. Note that OpenGL helper lanes are considered active, so if the
  1419. // program ever uses WQM, then the instruction and the first source will be
  1420. // computed in WQM.
  1421. def int_amdgcn_set_inactive :
  1422. Intrinsic<[llvm_anyint_ty],
  1423. [LLVMMatchType<0>, // value to be copied
  1424. LLVMMatchType<0>], // value for the inactive lanes to take
  1425. [IntrNoMem, IntrConvergent, IntrWillReturn]>;
  1426. // Return if the given flat pointer points to a local memory address.
  1427. def int_amdgcn_is_shared : GCCBuiltin<"__builtin_amdgcn_is_shared">,
  1428. Intrinsic<[llvm_i1_ty], [llvm_ptr_ty],
  1429. [IntrNoMem, IntrSpeculatable, NoCapture<ArgIndex<0>>, IntrWillReturn]
  1430. >;
  1431. // Return if the given flat pointer points to a prvate memory address.
  1432. def int_amdgcn_is_private : GCCBuiltin<"__builtin_amdgcn_is_private">,
  1433. Intrinsic<[llvm_i1_ty], [llvm_ptr_ty],
  1434. [IntrNoMem, IntrSpeculatable, NoCapture<ArgIndex<0>>, IntrWillReturn]
  1435. >;
  1436. //===----------------------------------------------------------------------===//
  1437. // CI+ Intrinsics
  1438. //===----------------------------------------------------------------------===//
  1439. def int_amdgcn_s_dcache_inv_vol :
  1440. GCCBuiltin<"__builtin_amdgcn_s_dcache_inv_vol">,
  1441. Intrinsic<[], [], [IntrNoMem, IntrHasSideEffects, IntrWillReturn]>;
  1442. def int_amdgcn_buffer_wbinvl1_vol :
  1443. GCCBuiltin<"__builtin_amdgcn_buffer_wbinvl1_vol">,
  1444. Intrinsic<[], [], [IntrNoMem, IntrHasSideEffects, IntrWillReturn]>;
  1445. //===----------------------------------------------------------------------===//
  1446. // VI Intrinsics
  1447. //===----------------------------------------------------------------------===//
  1448. // llvm.amdgcn.mov.dpp.i32 <src> <dpp_ctrl> <row_mask> <bank_mask> <bound_ctrl>
  1449. def int_amdgcn_mov_dpp :
  1450. Intrinsic<[llvm_anyint_ty],
  1451. [LLVMMatchType<0>, llvm_i32_ty, llvm_i32_ty, llvm_i32_ty,
  1452. llvm_i1_ty],
  1453. [IntrNoMem, IntrConvergent, IntrWillReturn,
  1454. ImmArg<ArgIndex<1>>, ImmArg<ArgIndex<2>>,
  1455. ImmArg<ArgIndex<3>>, ImmArg<ArgIndex<4>>]>;
  1456. // llvm.amdgcn.update.dpp.i32 <old> <src> <dpp_ctrl> <row_mask> <bank_mask> <bound_ctrl>
  1457. // Should be equivalent to:
  1458. // v_mov_b32 <dest> <old>
  1459. // v_mov_b32 <dest> <src> <dpp_ctrl> <row_mask> <bank_mask> <bound_ctrl>
  1460. def int_amdgcn_update_dpp :
  1461. Intrinsic<[llvm_anyint_ty],
  1462. [LLVMMatchType<0>, LLVMMatchType<0>, llvm_i32_ty,
  1463. llvm_i32_ty, llvm_i32_ty, llvm_i1_ty],
  1464. [IntrNoMem, IntrConvergent, IntrWillReturn,
  1465. ImmArg<ArgIndex<2>>, ImmArg<ArgIndex<3>>,
  1466. ImmArg<ArgIndex<4>>, ImmArg<ArgIndex<5>>]>;
  1467. def int_amdgcn_s_dcache_wb :
  1468. GCCBuiltin<"__builtin_amdgcn_s_dcache_wb">,
  1469. Intrinsic<[], [], [IntrNoMem, IntrHasSideEffects, IntrWillReturn]>;
  1470. def int_amdgcn_s_dcache_wb_vol :
  1471. GCCBuiltin<"__builtin_amdgcn_s_dcache_wb_vol">,
  1472. Intrinsic<[], [], [IntrNoMem, IntrHasSideEffects, IntrWillReturn]>;
  1473. def int_amdgcn_s_memrealtime :
  1474. GCCBuiltin<"__builtin_amdgcn_s_memrealtime">,
  1475. Intrinsic<[llvm_i64_ty], [], [IntrWillReturn]>;
  1476. // llvm.amdgcn.ds.permute <index> <src>
  1477. def int_amdgcn_ds_permute :
  1478. GCCBuiltin<"__builtin_amdgcn_ds_permute">,
  1479. Intrinsic<[llvm_i32_ty], [llvm_i32_ty, llvm_i32_ty],
  1480. [IntrNoMem, IntrConvergent, IntrWillReturn]>;
  1481. // llvm.amdgcn.ds.bpermute <index> <src>
  1482. def int_amdgcn_ds_bpermute :
  1483. GCCBuiltin<"__builtin_amdgcn_ds_bpermute">,
  1484. Intrinsic<[llvm_i32_ty], [llvm_i32_ty, llvm_i32_ty],
  1485. [IntrNoMem, IntrConvergent, IntrWillReturn]>;
  1486. // llvm.amdgcn.perm <src0> <src1> <selector>
  1487. def int_amdgcn_perm :
  1488. GCCBuiltin<"__builtin_amdgcn_perm">,
  1489. Intrinsic<[llvm_i32_ty], [llvm_i32_ty, llvm_i32_ty, llvm_i32_ty],
  1490. [IntrNoMem, IntrSpeculatable, IntrWillReturn]>;
  1491. //===----------------------------------------------------------------------===//
  1492. // GFX10 Intrinsics
  1493. //===----------------------------------------------------------------------===//
  1494. // llvm.amdgcn.permlane16 <old> <src0> <src1> <src2> <fi> <bound_control>
  1495. def int_amdgcn_permlane16 : GCCBuiltin<"__builtin_amdgcn_permlane16">,
  1496. Intrinsic<[llvm_i32_ty],
  1497. [llvm_i32_ty, llvm_i32_ty, llvm_i32_ty, llvm_i32_ty, llvm_i1_ty, llvm_i1_ty],
  1498. [IntrNoMem, IntrConvergent, IntrWillReturn,
  1499. ImmArg<ArgIndex<4>>, ImmArg<ArgIndex<5>>]>;
  1500. // llvm.amdgcn.permlanex16 <old> <src0> <src1> <src2> <fi> <bound_control>
  1501. def int_amdgcn_permlanex16 : GCCBuiltin<"__builtin_amdgcn_permlanex16">,
  1502. Intrinsic<[llvm_i32_ty],
  1503. [llvm_i32_ty, llvm_i32_ty, llvm_i32_ty, llvm_i32_ty, llvm_i1_ty, llvm_i1_ty],
  1504. [IntrNoMem, IntrConvergent, IntrWillReturn,
  1505. ImmArg<ArgIndex<4>>, ImmArg<ArgIndex<5>>]>;
  1506. // llvm.amdgcn.mov.dpp8.i32 <src> <sel>
  1507. // <sel> is a 32-bit constant whose high 8 bits must be zero which selects
  1508. // the lanes to read from.
  1509. def int_amdgcn_mov_dpp8 :
  1510. Intrinsic<[llvm_anyint_ty],
  1511. [LLVMMatchType<0>, llvm_i32_ty],
  1512. [IntrNoMem, IntrConvergent, IntrWillReturn,
  1513. ImmArg<ArgIndex<1>>]>;
  1514. def int_amdgcn_s_get_waveid_in_workgroup :
  1515. GCCBuiltin<"__builtin_amdgcn_s_get_waveid_in_workgroup">,
  1516. Intrinsic<[llvm_i32_ty], [],
  1517. [IntrReadMem, IntrInaccessibleMemOnly, IntrWillReturn]>;
  1518. class AMDGPUGlobalAtomicRtn<LLVMType vt> : Intrinsic <
  1519. [vt],
  1520. [llvm_anyptr_ty, // vaddr
  1521. vt], // vdata(VGPR)
  1522. [IntrArgMemOnly, IntrWillReturn, NoCapture<ArgIndex<0>>], "",
  1523. [SDNPMemOperand]>;
  1524. def int_amdgcn_global_atomic_csub : AMDGPUGlobalAtomicRtn<llvm_i32_ty>;
  1525. // uint4 llvm.amdgcn.image.bvh.intersect.ray <node_ptr>, <ray_extent>, <ray_origin>,
  1526. // <ray_dir>, <ray_inv_dir>, <texture_descr>
  1527. def int_amdgcn_image_bvh_intersect_ray :
  1528. Intrinsic<[llvm_v4i32_ty],
  1529. [llvm_anyint_ty, llvm_float_ty, llvm_v4f32_ty, llvm_anyvector_ty,
  1530. LLVMMatchType<1>, llvm_v4i32_ty],
  1531. [IntrReadMem, IntrWillReturn]>;
  1532. //===----------------------------------------------------------------------===//
  1533. // Deep learning intrinsics.
  1534. //===----------------------------------------------------------------------===//
  1535. // f32 %r = llvm.amdgcn.fdot2(v2f16 %a, v2f16 %b, f32 %c, i1 %clamp)
  1536. // %r = %a[0] * %b[0] + %a[1] * %b[1] + %c
  1537. def int_amdgcn_fdot2 :
  1538. GCCBuiltin<"__builtin_amdgcn_fdot2">,
  1539. Intrinsic<
  1540. [llvm_float_ty], // %r
  1541. [
  1542. llvm_v2f16_ty, // %a
  1543. llvm_v2f16_ty, // %b
  1544. llvm_float_ty, // %c
  1545. llvm_i1_ty // %clamp
  1546. ],
  1547. [IntrNoMem, IntrSpeculatable, IntrWillReturn, ImmArg<ArgIndex<3>>]
  1548. >;
  1549. // i32 %r = llvm.amdgcn.sdot2(v2i16 %a, v2i16 %b, i32 %c, i1 %clamp)
  1550. // %r = %a[0] * %b[0] + %a[1] * %b[1] + %c
  1551. def int_amdgcn_sdot2 :
  1552. GCCBuiltin<"__builtin_amdgcn_sdot2">,
  1553. Intrinsic<
  1554. [llvm_i32_ty], // %r
  1555. [
  1556. llvm_v2i16_ty, // %a
  1557. llvm_v2i16_ty, // %b
  1558. llvm_i32_ty, // %c
  1559. llvm_i1_ty // %clamp
  1560. ],
  1561. [IntrNoMem, IntrSpeculatable, IntrWillReturn, ImmArg<ArgIndex<3>>]
  1562. >;
  1563. // u32 %r = llvm.amdgcn.udot2(v2u16 %a, v2u16 %b, u32 %c, i1 %clamp)
  1564. // %r = %a[0] * %b[0] + %a[1] * %b[1] + %c
  1565. def int_amdgcn_udot2 :
  1566. GCCBuiltin<"__builtin_amdgcn_udot2">,
  1567. Intrinsic<
  1568. [llvm_i32_ty], // %r
  1569. [
  1570. llvm_v2i16_ty, // %a
  1571. llvm_v2i16_ty, // %b
  1572. llvm_i32_ty, // %c
  1573. llvm_i1_ty // %clamp
  1574. ],
  1575. [IntrNoMem, IntrSpeculatable, IntrWillReturn, ImmArg<ArgIndex<3>>]
  1576. >;
  1577. // i32 %r = llvm.amdgcn.sdot4(v4i8 (as i32) %a, v4i8 (as i32) %b, i32 %c, i1 %clamp)
  1578. // %r = %a[0] * %b[0] + %a[1] * %b[1] + %a[2] * %b[2] + %a[3] * %b[3] + %c
  1579. def int_amdgcn_sdot4 :
  1580. GCCBuiltin<"__builtin_amdgcn_sdot4">,
  1581. Intrinsic<
  1582. [llvm_i32_ty], // %r
  1583. [
  1584. llvm_i32_ty, // %a
  1585. llvm_i32_ty, // %b
  1586. llvm_i32_ty, // %c
  1587. llvm_i1_ty // %clamp
  1588. ],
  1589. [IntrNoMem, IntrSpeculatable, IntrWillReturn, ImmArg<ArgIndex<3>>]
  1590. >;
  1591. // u32 %r = llvm.amdgcn.udot4(v4u8 (as u32) %a, v4u8 (as u32) %b, u32 %c, i1 %clamp)
  1592. // %r = %a[0] * %b[0] + %a[1] * %b[1] + %a[2] * %b[2] + %a[3] * %b[3] + %c
  1593. def int_amdgcn_udot4 :
  1594. GCCBuiltin<"__builtin_amdgcn_udot4">,
  1595. Intrinsic<
  1596. [llvm_i32_ty], // %r
  1597. [
  1598. llvm_i32_ty, // %a
  1599. llvm_i32_ty, // %b
  1600. llvm_i32_ty, // %c
  1601. llvm_i1_ty // %clamp
  1602. ],
  1603. [IntrNoMem, IntrSpeculatable, IntrWillReturn, ImmArg<ArgIndex<3>>]
  1604. >;
  1605. // i32 %r = llvm.amdgcn.sdot8(v8i4 (as i32) %a, v8i4 (as i32) %b, i32 %c, i1 %clamp)
  1606. // %r = %a[0] * %b[0] + %a[1] * %b[1] + %a[2] * %b[2] + %a[3] * %b[3] +
  1607. // %a[4] * %b[4] + %a[5] * %b[5] + %a[6] * %b[6] + %a[7] * %b[7] + %c
  1608. def int_amdgcn_sdot8 :
  1609. GCCBuiltin<"__builtin_amdgcn_sdot8">,
  1610. Intrinsic<
  1611. [llvm_i32_ty], // %r
  1612. [
  1613. llvm_i32_ty, // %a
  1614. llvm_i32_ty, // %b
  1615. llvm_i32_ty, // %c
  1616. llvm_i1_ty // %clamp
  1617. ],
  1618. [IntrNoMem, IntrSpeculatable, IntrWillReturn, ImmArg<ArgIndex<3>>]
  1619. >;
  1620. // u32 %r = llvm.amdgcn.udot8(v8u4 (as u32) %a, v8u4 (as u32) %b, u32 %c, i1 %clamp)
  1621. // %r = %a[0] * %b[0] + %a[1] * %b[1] + %a[2] * %b[2] + %a[3] * %b[3] +
  1622. // %a[4] * %b[4] + %a[5] * %b[5] + %a[6] * %b[6] + %a[7] * %b[7] + %c
  1623. def int_amdgcn_udot8 :
  1624. GCCBuiltin<"__builtin_amdgcn_udot8">,
  1625. Intrinsic<
  1626. [llvm_i32_ty], // %r
  1627. [
  1628. llvm_i32_ty, // %a
  1629. llvm_i32_ty, // %b
  1630. llvm_i32_ty, // %c
  1631. llvm_i1_ty // %clamp
  1632. ],
  1633. [IntrNoMem, IntrSpeculatable, IntrWillReturn, ImmArg<ArgIndex<3>>]
  1634. >;
  1635. //===----------------------------------------------------------------------===//
  1636. // gfx908 intrinsics
  1637. // ===----------------------------------------------------------------------===//
  1638. def int_amdgcn_global_atomic_fadd : AMDGPUGlobalAtomicRtn<llvm_anyfloat_ty>;
  1639. // llvm.amdgcn.mfma.*.* vdst, srcA, srcB, srcC, cbsz, abid, blgp
  1640. class AMDGPUMfmaIntrinsic<LLVMType DestTy, LLVMType SrcABTy> :
  1641. GCCBuiltin<!subst("int", "__builtin", NAME)>,
  1642. Intrinsic<[DestTy],
  1643. [SrcABTy, SrcABTy, DestTy,
  1644. llvm_i32_ty, llvm_i32_ty, llvm_i32_ty],
  1645. [IntrConvergent, IntrNoMem, IntrWillReturn,
  1646. ImmArg<ArgIndex<3>>, ImmArg<ArgIndex<4>>, ImmArg<ArgIndex<5>>]>;
  1647. def int_amdgcn_mfma_f32_32x32x1f32 : AMDGPUMfmaIntrinsic<llvm_v32f32_ty, llvm_float_ty>;
  1648. def int_amdgcn_mfma_f32_16x16x1f32 : AMDGPUMfmaIntrinsic<llvm_v16f32_ty, llvm_float_ty>;
  1649. def int_amdgcn_mfma_f32_4x4x1f32 : AMDGPUMfmaIntrinsic<llvm_v4f32_ty, llvm_float_ty>;
  1650. def int_amdgcn_mfma_f32_32x32x2f32 : AMDGPUMfmaIntrinsic<llvm_v16f32_ty, llvm_float_ty>;
  1651. def int_amdgcn_mfma_f32_16x16x4f32 : AMDGPUMfmaIntrinsic<llvm_v4f32_ty, llvm_float_ty>;
  1652. def int_amdgcn_mfma_f32_32x32x4f16 : AMDGPUMfmaIntrinsic<llvm_v32f32_ty, llvm_v4f16_ty>;
  1653. def int_amdgcn_mfma_f32_16x16x4f16 : AMDGPUMfmaIntrinsic<llvm_v16f32_ty, llvm_v4f16_ty>;
  1654. def int_amdgcn_mfma_f32_4x4x4f16 : AMDGPUMfmaIntrinsic<llvm_v4f32_ty, llvm_v4f16_ty>;
  1655. def int_amdgcn_mfma_f32_32x32x8f16 : AMDGPUMfmaIntrinsic<llvm_v16f32_ty, llvm_v4f16_ty>;
  1656. def int_amdgcn_mfma_f32_16x16x16f16 : AMDGPUMfmaIntrinsic<llvm_v4f32_ty, llvm_v4f16_ty>;
  1657. def int_amdgcn_mfma_i32_32x32x4i8 : AMDGPUMfmaIntrinsic<llvm_v32i32_ty, llvm_i32_ty>;
  1658. def int_amdgcn_mfma_i32_16x16x4i8 : AMDGPUMfmaIntrinsic<llvm_v16i32_ty, llvm_i32_ty>;
  1659. def int_amdgcn_mfma_i32_4x4x4i8 : AMDGPUMfmaIntrinsic<llvm_v4i32_ty, llvm_i32_ty>;
  1660. def int_amdgcn_mfma_i32_32x32x8i8 : AMDGPUMfmaIntrinsic<llvm_v16i32_ty, llvm_i32_ty>;
  1661. def int_amdgcn_mfma_i32_16x16x16i8 : AMDGPUMfmaIntrinsic<llvm_v4i32_ty, llvm_i32_ty>;
  1662. def int_amdgcn_mfma_f32_32x32x2bf16 : AMDGPUMfmaIntrinsic<llvm_v32f32_ty, llvm_v2i16_ty>;
  1663. def int_amdgcn_mfma_f32_16x16x2bf16 : AMDGPUMfmaIntrinsic<llvm_v16f32_ty, llvm_v2i16_ty>;
  1664. def int_amdgcn_mfma_f32_4x4x2bf16 : AMDGPUMfmaIntrinsic<llvm_v4f32_ty, llvm_v2i16_ty>;
  1665. def int_amdgcn_mfma_f32_32x32x4bf16 : AMDGPUMfmaIntrinsic<llvm_v16f32_ty, llvm_v2i16_ty>;
  1666. def int_amdgcn_mfma_f32_16x16x8bf16 : AMDGPUMfmaIntrinsic<llvm_v4f32_ty, llvm_v2i16_ty>;
  1667. //===----------------------------------------------------------------------===//
  1668. // gfx90a intrinsics
  1669. // ===----------------------------------------------------------------------===//
  1670. def int_amdgcn_global_atomic_fmin : AMDGPUGlobalAtomicRtn<llvm_anyfloat_ty>;
  1671. def int_amdgcn_global_atomic_fmax : AMDGPUGlobalAtomicRtn<llvm_anyfloat_ty>;
  1672. def int_amdgcn_flat_atomic_fadd : AMDGPUGlobalAtomicRtn<llvm_anyfloat_ty>;
  1673. def int_amdgcn_flat_atomic_fmin : AMDGPUGlobalAtomicRtn<llvm_anyfloat_ty>;
  1674. def int_amdgcn_flat_atomic_fmax : AMDGPUGlobalAtomicRtn<llvm_anyfloat_ty>;
  1675. def int_amdgcn_mfma_f32_32x32x4bf16_1k : AMDGPUMfmaIntrinsic<llvm_v32f32_ty, llvm_v4i16_ty>;
  1676. def int_amdgcn_mfma_f32_16x16x4bf16_1k : AMDGPUMfmaIntrinsic<llvm_v16f32_ty, llvm_v4i16_ty>;
  1677. def int_amdgcn_mfma_f32_4x4x4bf16_1k : AMDGPUMfmaIntrinsic<llvm_v4f32_ty, llvm_v4i16_ty>;
  1678. def int_amdgcn_mfma_f32_32x32x8bf16_1k : AMDGPUMfmaIntrinsic<llvm_v16f32_ty, llvm_v4i16_ty>;
  1679. def int_amdgcn_mfma_f32_16x16x16bf16_1k : AMDGPUMfmaIntrinsic<llvm_v4f32_ty, llvm_v4i16_ty>;
  1680. def int_amdgcn_mfma_f64_16x16x4f64 : AMDGPUMfmaIntrinsic<llvm_v4f64_ty, llvm_double_ty>;
  1681. def int_amdgcn_mfma_f64_4x4x4f64 : AMDGPUMfmaIntrinsic<llvm_double_ty, llvm_double_ty>;
  1682. //===----------------------------------------------------------------------===//
  1683. // Special Intrinsics for backend internal use only. No frontend
  1684. // should emit calls to these.
  1685. // ===----------------------------------------------------------------------===//
  1686. def int_amdgcn_if : Intrinsic<[llvm_i1_ty, llvm_anyint_ty],
  1687. [llvm_i1_ty], [IntrConvergent, IntrWillReturn]
  1688. >;
  1689. def int_amdgcn_else : Intrinsic<[llvm_i1_ty, llvm_anyint_ty],
  1690. [llvm_anyint_ty], [IntrConvergent, IntrWillReturn]
  1691. >;
  1692. def int_amdgcn_if_break : Intrinsic<[llvm_anyint_ty],
  1693. [llvm_i1_ty, LLVMMatchType<0>],
  1694. [IntrNoMem, IntrConvergent, IntrWillReturn]
  1695. >;
  1696. def int_amdgcn_loop : Intrinsic<[llvm_i1_ty],
  1697. [llvm_anyint_ty], [IntrConvergent, IntrWillReturn]
  1698. >;
  1699. def int_amdgcn_end_cf : Intrinsic<[], [llvm_anyint_ty],
  1700. [IntrConvergent, IntrWillReturn]>;
  1701. // Represent unreachable in a divergent region.
  1702. def int_amdgcn_unreachable : Intrinsic<[], [], [IntrConvergent]>;
  1703. // Emit 2.5 ulp, no denormal division. Should only be inserted by
  1704. // pass based on !fpmath metadata.
  1705. def int_amdgcn_fdiv_fast : Intrinsic<
  1706. [llvm_float_ty], [llvm_float_ty, llvm_float_ty],
  1707. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  1708. >;
  1709. // Represent a relocation constant.
  1710. def int_amdgcn_reloc_constant : Intrinsic<
  1711. [llvm_i32_ty], [llvm_metadata_ty],
  1712. [IntrNoMem, IntrSpeculatable, IntrWillReturn]
  1713. >;
  1714. }