| 1 | //! This file is auto-generated by tools/update_cpu_features.zig. |
| 2 | |
| 3 | const std = @import("../std.zig"); |
| 4 | const CpuFeature = std.Target.Cpu.Feature; |
| 5 | const CpuModel = std.Target.Cpu.Model; |
| 6 | |
| 7 | pub const Feature = enum { |
| 8 | @"1024_addressable_vgprs", |
| 9 | @"16_bit_insts", |
| 10 | @"45_bit_num_records_buffer_resource", |
| 11 | @"64_bit_literals", |
| 12 | a16, |
| 13 | add_min_max_insts, |
| 14 | add_no_carry_insts, |
| 15 | add_sub_u64_insts, |
| 16 | addressablelocalmemorysize163840, |
| 17 | addressablelocalmemorysize32768, |
| 18 | addressablelocalmemorysize327680, |
| 19 | addressablelocalmemorysize65536, |
| 20 | agent_scope_fine_grained_remote_memory_atomics, |
| 21 | allocate1_5xvgprs, |
| 22 | aperture_regs, |
| 23 | architected_flat_scratch, |
| 24 | architected_sgprs, |
| 25 | ashr_pk_insts, |
| 26 | assembler_permissive_wavesize, |
| 27 | atomic_buffer_global_pk_add_f16_insts, |
| 28 | atomic_buffer_global_pk_add_f16_no_rtn_insts, |
| 29 | atomic_buffer_pk_add_bf16_inst, |
| 30 | atomic_csub_no_rtn_insts, |
| 31 | atomic_ds_pk_add_16_insts, |
| 32 | atomic_fadd_no_rtn_insts, |
| 33 | atomic_fadd_rtn_insts, |
| 34 | atomic_flat_pk_add_16_insts, |
| 35 | atomic_fmin_fmax_flat_f32, |
| 36 | atomic_fmin_fmax_flat_f64, |
| 37 | atomic_fmin_fmax_global_f32, |
| 38 | atomic_fmin_fmax_global_f64, |
| 39 | atomic_global_pk_add_bf16_inst, |
| 40 | auto_waitcnt_before_barrier, |
| 41 | back_off_barrier, |
| 42 | bf16_cvt_insts, |
| 43 | bf16_pk_insts, |
| 44 | bf16_trans_insts, |
| 45 | bf8_cvt_scale_insts, |
| 46 | bitop3_insts, |
| 47 | block_vgpr_csr, |
| 48 | bvh_dual_bvh_8_insts, |
| 49 | ci_insts, |
| 50 | clusters, |
| 51 | cube_insts, |
| 52 | cumode, |
| 53 | cvt_fp8_vop1_bug, |
| 54 | cvt_norm_insts, |
| 55 | cvt_pk_f16_f32_inst, |
| 56 | cvt_pknorm_vop2_insts, |
| 57 | cvt_pknorm_vop3_insts, |
| 58 | d16_write_vgpr32, |
| 59 | default_component_broadcast, |
| 60 | default_component_zero, |
| 61 | dl_insts, |
| 62 | dot10_insts, |
| 63 | dot11_insts, |
| 64 | dot12_insts, |
| 65 | dot13_insts, |
| 66 | dot1_insts, |
| 67 | dot2_insts, |
| 68 | dot3_insts, |
| 69 | dot4_insts, |
| 70 | dot5_insts, |
| 71 | dot6_insts, |
| 72 | dot7_insts, |
| 73 | dot8_insts, |
| 74 | dot9_insts, |
| 75 | dpp, |
| 76 | dpp8, |
| 77 | dpp_64bit, |
| 78 | dpp_src1_sgpr, |
| 79 | ds128, |
| 80 | ds_src2_insts, |
| 81 | emulated_system_scope_atomics, |
| 82 | extended_image_insts, |
| 83 | f16bf16_to_fp6bf6_cvt_scale_insts, |
| 84 | f32_to_f16bf16_cvt_sr_insts, |
| 85 | fast_denormal_f32, |
| 86 | fast_fmaf, |
| 87 | flat_address_space, |
| 88 | flat_atomic_fadd_f32_inst, |
| 89 | flat_buffer_global_fadd_f64_inst, |
| 90 | flat_for_global, |
| 91 | flat_global_insts, |
| 92 | flat_gvs_mode, |
| 93 | flat_inst_offsets, |
| 94 | flat_scratch, |
| 95 | flat_scratch_insts, |
| 96 | flat_segment_offset_bug, |
| 97 | fma_mix_bf16_insts, |
| 98 | fma_mix_insts, |
| 99 | fmacf64_inst, |
| 100 | fmaf, |
| 101 | fp4_cvt_scale_insts, |
| 102 | fp64, |
| 103 | fp6bf6_cvt_scale_insts, |
| 104 | fp8_conversion_insts, |
| 105 | fp8_cvt_scale_insts, |
| 106 | fp8_insts, |
| 107 | fp8e5m3_insts, |
| 108 | full_rate_64_ops, |
| 109 | g16, |
| 110 | gcn3_encoding, |
| 111 | gds, |
| 112 | get_wave_id_inst, |
| 113 | gfx10, |
| 114 | gfx10_3_insts, |
| 115 | gfx10_a_encoding, |
| 116 | gfx10_b_encoding, |
| 117 | gfx10_insts, |
| 118 | gfx11, |
| 119 | gfx11_insts, |
| 120 | gfx12, |
| 121 | gfx1250_insts, |
| 122 | gfx12_insts, |
| 123 | gfx7_gfx8_gfx9_insts, |
| 124 | gfx8_insts, |
| 125 | gfx9, |
| 126 | gfx90a_insts, |
| 127 | gfx940_insts, |
| 128 | gfx950_insts, |
| 129 | gfx9_insts, |
| 130 | globally_addressable_scratch, |
| 131 | gws, |
| 132 | half_rate_64_ops, |
| 133 | ieee_minimum_maximum_insts, |
| 134 | image_gather4_d16_bug, |
| 135 | image_insts, |
| 136 | image_store_d16_bug, |
| 137 | inst_fwd_prefetch_bug, |
| 138 | int_clamp_insts, |
| 139 | inv_2pi_inline_imm, |
| 140 | kernarg_preload, |
| 141 | lds_barrier_arrive_atomic, |
| 142 | lds_branch_vmem_war_hazard, |
| 143 | lds_misaligned_bug, |
| 144 | ldsbankcount16, |
| 145 | ldsbankcount32, |
| 146 | lerp_inst, |
| 147 | load_store_opt, |
| 148 | lshl_add_u64_inst, |
| 149 | mad_intra_fwd_bug, |
| 150 | mad_mac_f32_insts, |
| 151 | mad_mix_insts, |
| 152 | mad_u32_inst, |
| 153 | mai_insts, |
| 154 | max_hard_clause_length_32, |
| 155 | max_hard_clause_length_63, |
| 156 | max_private_element_size_16, |
| 157 | max_private_element_size_4, |
| 158 | max_private_element_size_8, |
| 159 | mcast_load_insts, |
| 160 | memory_atomic_fadd_f32_denormal_support, |
| 161 | mfma_inline_literal_bug, |
| 162 | mimg_r128, |
| 163 | min3_max3_pkf16, |
| 164 | minimum3_maximum3_f16, |
| 165 | minimum3_maximum3_f32, |
| 166 | minimum3_maximum3_pkf16, |
| 167 | movrel, |
| 168 | msaa_load_dst_sel_bug, |
| 169 | negative_scratch_offset_bug, |
| 170 | negative_unaligned_scratch_offset_bug, |
| 171 | no_data_dep_hazard, |
| 172 | no_sdst_cmpx, |
| 173 | nsa_clause_bug, |
| 174 | nsa_encoding, |
| 175 | nsa_to_vmem_bug, |
| 176 | offset_3f_bug, |
| 177 | packed_fp32_ops, |
| 178 | packed_tid, |
| 179 | partial_nsa_encoding, |
| 180 | permlane16_swap, |
| 181 | permlane32_swap, |
| 182 | pk_add_min_max_insts, |
| 183 | pk_fmac_f16_inst, |
| 184 | point_sample_accel, |
| 185 | precise_memory, |
| 186 | priv_enabled_trap2_nop_bug, |
| 187 | prng_inst, |
| 188 | promote_alloca, |
| 189 | prt_strict_null, |
| 190 | pseudo_scalar_trans, |
| 191 | qsad_insts, |
| 192 | r128_a16, |
| 193 | real_true16, |
| 194 | relaxed_buffer_oob_mode, |
| 195 | required_export_priority, |
| 196 | requires_cov6, |
| 197 | restricted_soffset, |
| 198 | s_memrealtime, |
| 199 | s_memtime_inst, |
| 200 | s_wakeup_barrier_inst, |
| 201 | sad_insts, |
| 202 | safe_cu_prefetch, |
| 203 | safe_smem_prefetch, |
| 204 | salu_float, |
| 205 | scalar_atomics, |
| 206 | scalar_dwordx3_loads, |
| 207 | scalar_flat_scratch_insts, |
| 208 | scalar_stores, |
| 209 | sdwa, |
| 210 | sdwa_mav, |
| 211 | sdwa_omod, |
| 212 | sdwa_out_mods_vopc, |
| 213 | sdwa_scalar, |
| 214 | sdwa_sdst, |
| 215 | sea_islands, |
| 216 | setprio_inc_wg_inst, |
| 217 | setreg_vgpr_msb_fixup, |
| 218 | sgpr_init_bug, |
| 219 | shader_cycles_hi_lo_registers, |
| 220 | shader_cycles_register, |
| 221 | si_scheduler, |
| 222 | smem_to_vector_write_hazard, |
| 223 | southern_islands, |
| 224 | sramecc, |
| 225 | sramecc_support, |
| 226 | tanh_insts, |
| 227 | tensor_cvt_lut_insts, |
| 228 | tgsplit, |
| 229 | transpose_load_f4f6_insts, |
| 230 | trap_handler, |
| 231 | trig_reduced_range, |
| 232 | true16, |
| 233 | unaligned_access_mode, |
| 234 | unaligned_buffer_access, |
| 235 | unaligned_ds_access, |
| 236 | unaligned_scratch_access, |
| 237 | unpacked_d16_vmem, |
| 238 | unsafe_ds_offset_folding, |
| 239 | user_sgpr_init16_bug, |
| 240 | valu_trans_use_hazard, |
| 241 | vcmpx_exec_war_hazard, |
| 242 | vcmpx_permlane_hazard, |
| 243 | vgpr_align2, |
| 244 | vgpr_index_mode, |
| 245 | vmem_pref_insts, |
| 246 | vmem_to_lds_load_insts, |
| 247 | vmem_to_scalar_write_hazard, |
| 248 | vmem_write_vgpr_in_order, |
| 249 | volcanic_islands, |
| 250 | vop3_literal, |
| 251 | vop3p, |
| 252 | vopd, |
| 253 | vscnt, |
| 254 | wait_xcnt, |
| 255 | waits_before_system_scope_stores, |
| 256 | wavefrontsize16, |
| 257 | wavefrontsize32, |
| 258 | wavefrontsize64, |
| 259 | xf32_insts, |
| 260 | xnack, |
| 261 | xnack_support, |
| 262 | }; |
| 263 | |
| 264 | pub const featureSet = CpuFeature.FeatureSetFns(Feature).featureSet; |
| 265 | pub const featureSetHas = CpuFeature.FeatureSetFns(Feature).featureSetHas; |
| 266 | pub const featureSetHasAny = CpuFeature.FeatureSetFns(Feature).featureSetHasAny; |
| 267 | pub const featureSetHasAll = CpuFeature.FeatureSetFns(Feature).featureSetHasAll; |
| 268 | |
| 269 | pub const all_features = blk: { |
| 270 | @setEvalBranchQuota(2000); |
| 271 | const len = @typeInfo(Feature).@"enum".field_names.len; |
| 272 | std.debug.assert(len <= CpuFeature.Set.needed_bit_count); |
| 273 | var result: [len]CpuFeature = undefined; |
| 274 | result[@backingInt(Feature.@"1024_addressable_vgprs")] = .{ |
| 275 | .llvm_name = "1024-addressable-vgprs", |
| 276 | .description = "Has 1024 addressable VGPRs", |
| 277 | .dependencies = featureSet(&[_]Feature{}), |
| 278 | }; |
| 279 | result[@backingInt(Feature.@"16_bit_insts")] = .{ |
| 280 | .llvm_name = "16-bit-insts", |
| 281 | .description = "Has i16/f16 instructions", |
| 282 | .dependencies = featureSet(&[_]Feature{}), |
| 283 | }; |
| 284 | result[@backingInt(Feature.@"45_bit_num_records_buffer_resource")] = .{ |
| 285 | .llvm_name = "45-bit-num-records-buffer-resource", |
| 286 | .description = "The buffer resource (V#) supports 45-bit num_records", |
| 287 | .dependencies = featureSet(&[_]Feature{}), |
| 288 | }; |
| 289 | result[@backingInt(Feature.@"64_bit_literals")] = .{ |
| 290 | .llvm_name = "64-bit-literals", |
| 291 | .description = "Can use 64-bit literals with single DWORD instructions", |
| 292 | .dependencies = featureSet(&[_]Feature{}), |
| 293 | }; |
| 294 | result[@backingInt(Feature.a16)] = .{ |
| 295 | .llvm_name = "a16", |
| 296 | .description = "Support A16 for 16-bit coordinates/gradients/lod/clamp/mip image operands", |
| 297 | .dependencies = featureSet(&[_]Feature{}), |
| 298 | }; |
| 299 | result[@backingInt(Feature.add_min_max_insts)] = .{ |
| 300 | .llvm_name = "add-min-max-insts", |
| 301 | .description = "Has v_add_{min|max}_{i|u}32 instructions", |
| 302 | .dependencies = featureSet(&[_]Feature{}), |
| 303 | }; |
| 304 | result[@backingInt(Feature.add_no_carry_insts)] = .{ |
| 305 | .llvm_name = "add-no-carry-insts", |
| 306 | .description = "Have VALU add/sub instructions without carry out", |
| 307 | .dependencies = featureSet(&[_]Feature{}), |
| 308 | }; |
| 309 | result[@backingInt(Feature.add_sub_u64_insts)] = .{ |
| 310 | .llvm_name = "add-sub-u64-insts", |
| 311 | .description = "Has v_add_u64 and v_sub_u64 instructions", |
| 312 | .dependencies = featureSet(&[_]Feature{}), |
| 313 | }; |
| 314 | result[@backingInt(Feature.addressablelocalmemorysize163840)] = .{ |
| 315 | .llvm_name = "addressablelocalmemorysize163840", |
| 316 | .description = "The size of local memory in bytes", |
| 317 | .dependencies = featureSet(&[_]Feature{}), |
| 318 | }; |
| 319 | result[@backingInt(Feature.addressablelocalmemorysize32768)] = .{ |
| 320 | .llvm_name = "addressablelocalmemorysize32768", |
| 321 | .description = "The size of local memory in bytes", |
| 322 | .dependencies = featureSet(&[_]Feature{}), |
| 323 | }; |
| 324 | result[@backingInt(Feature.addressablelocalmemorysize327680)] = .{ |
| 325 | .llvm_name = "addressablelocalmemorysize327680", |
| 326 | .description = "The size of local memory in bytes", |
| 327 | .dependencies = featureSet(&[_]Feature{}), |
| 328 | }; |
| 329 | result[@backingInt(Feature.addressablelocalmemorysize65536)] = .{ |
| 330 | .llvm_name = "addressablelocalmemorysize65536", |
| 331 | .description = "The size of local memory in bytes", |
| 332 | .dependencies = featureSet(&[_]Feature{}), |
| 333 | }; |
| 334 | result[@backingInt(Feature.agent_scope_fine_grained_remote_memory_atomics)] = .{ |
| 335 | .llvm_name = "agent-scope-fine-grained-remote-memory-atomics", |
| 336 | .description = "Agent (device) scoped atomic operations, excluding those directly supported by PCIe (i.e. integer atomic add, exchange, and compare-and-swap), are functional for allocations in host or peer device memory.", |
| 337 | .dependencies = featureSet(&[_]Feature{}), |
| 338 | }; |
| 339 | result[@backingInt(Feature.allocate1_5xvgprs)] = .{ |
| 340 | .llvm_name = "allocate1_5xvgprs", |
| 341 | .description = "Has 50% more physical VGPRs and 50% larger allocation granule", |
| 342 | .dependencies = featureSet(&[_]Feature{}), |
| 343 | }; |
| 344 | result[@backingInt(Feature.aperture_regs)] = .{ |
| 345 | .llvm_name = "aperture-regs", |
| 346 | .description = "Has Memory Aperture Base and Size Registers", |
| 347 | .dependencies = featureSet(&[_]Feature{}), |
| 348 | }; |
| 349 | result[@backingInt(Feature.architected_flat_scratch)] = .{ |
| 350 | .llvm_name = "architected-flat-scratch", |
| 351 | .description = "Flat Scratch register is a readonly SPI initialized architected register", |
| 352 | .dependencies = featureSet(&[_]Feature{}), |
| 353 | }; |
| 354 | result[@backingInt(Feature.architected_sgprs)] = .{ |
| 355 | .llvm_name = "architected-sgprs", |
| 356 | .description = "Enable the architected SGPRs", |
| 357 | .dependencies = featureSet(&[_]Feature{}), |
| 358 | }; |
| 359 | result[@backingInt(Feature.ashr_pk_insts)] = .{ |
| 360 | .llvm_name = "ashr-pk-insts", |
| 361 | .description = "Has Arithmetic Shift Pack instructions", |
| 362 | .dependencies = featureSet(&[_]Feature{}), |
| 363 | }; |
| 364 | result[@backingInt(Feature.assembler_permissive_wavesize)] = .{ |
| 365 | .llvm_name = "assembler-permissive-wavesize", |
| 366 | .description = "allow parsing wave32 and wave64 variants of instructions", |
| 367 | .dependencies = featureSet(&[_]Feature{}), |
| 368 | }; |
| 369 | result[@backingInt(Feature.atomic_buffer_global_pk_add_f16_insts)] = .{ |
| 370 | .llvm_name = "atomic-buffer-global-pk-add-f16-insts", |
| 371 | .description = "Has buffer_atomic_pk_add_f16 and global_atomic_pk_add_f16 instructions that can return original value", |
| 372 | .dependencies = featureSet(&[_]Feature{ |
| 373 | .flat_global_insts, |
| 374 | }), |
| 375 | }; |
| 376 | result[@backingInt(Feature.atomic_buffer_global_pk_add_f16_no_rtn_insts)] = .{ |
| 377 | .llvm_name = "atomic-buffer-global-pk-add-f16-no-rtn-insts", |
| 378 | .description = "Has buffer_atomic_pk_add_f16 and global_atomic_pk_add_f16 instructions that don't return original value", |
| 379 | .dependencies = featureSet(&[_]Feature{ |
| 380 | .flat_global_insts, |
| 381 | }), |
| 382 | }; |
| 383 | result[@backingInt(Feature.atomic_buffer_pk_add_bf16_inst)] = .{ |
| 384 | .llvm_name = "atomic-buffer-pk-add-bf16-inst", |
| 385 | .description = "Has buffer_atomic_pk_add_bf16 instruction", |
| 386 | .dependencies = featureSet(&[_]Feature{}), |
| 387 | }; |
| 388 | result[@backingInt(Feature.atomic_csub_no_rtn_insts)] = .{ |
| 389 | .llvm_name = "atomic-csub-no-rtn-insts", |
| 390 | .description = "Has buffer_atomic_csub and global_atomic_csub instructions that don't return original value", |
| 391 | .dependencies = featureSet(&[_]Feature{}), |
| 392 | }; |
| 393 | result[@backingInt(Feature.atomic_ds_pk_add_16_insts)] = .{ |
| 394 | .llvm_name = "atomic-ds-pk-add-16-insts", |
| 395 | .description = "Has ds_pk_add_bf16, ds_pk_add_f16, ds_pk_add_rtn_bf16, ds_pk_add_rtn_f16 instructions", |
| 396 | .dependencies = featureSet(&[_]Feature{}), |
| 397 | }; |
| 398 | result[@backingInt(Feature.atomic_fadd_no_rtn_insts)] = .{ |
| 399 | .llvm_name = "atomic-fadd-no-rtn-insts", |
| 400 | .description = "Has buffer_atomic_add_f32 and global_atomic_add_f32 instructions that don't return original value", |
| 401 | .dependencies = featureSet(&[_]Feature{ |
| 402 | .flat_global_insts, |
| 403 | }), |
| 404 | }; |
| 405 | result[@backingInt(Feature.atomic_fadd_rtn_insts)] = .{ |
| 406 | .llvm_name = "atomic-fadd-rtn-insts", |
| 407 | .description = "Has buffer_atomic_add_f32 and global_atomic_add_f32 instructions that return original value", |
| 408 | .dependencies = featureSet(&[_]Feature{ |
| 409 | .flat_global_insts, |
| 410 | }), |
| 411 | }; |
| 412 | result[@backingInt(Feature.atomic_flat_pk_add_16_insts)] = .{ |
| 413 | .llvm_name = "atomic-flat-pk-add-16-insts", |
| 414 | .description = "Has flat_atomic_pk_add_f16 and flat_atomic_pk_add_bf16 instructions", |
| 415 | .dependencies = featureSet(&[_]Feature{}), |
| 416 | }; |
| 417 | result[@backingInt(Feature.atomic_fmin_fmax_flat_f32)] = .{ |
| 418 | .llvm_name = "atomic-fmin-fmax-flat-f32", |
| 419 | .description = "Has flat memory instructions for atomicrmw fmin/fmax for float", |
| 420 | .dependencies = featureSet(&[_]Feature{ |
| 421 | .flat_address_space, |
| 422 | }), |
| 423 | }; |
| 424 | result[@backingInt(Feature.atomic_fmin_fmax_flat_f64)] = .{ |
| 425 | .llvm_name = "atomic-fmin-fmax-flat-f64", |
| 426 | .description = "Has flat memory instructions for atomicrmw fmin/fmax for double", |
| 427 | .dependencies = featureSet(&[_]Feature{ |
| 428 | .flat_address_space, |
| 429 | }), |
| 430 | }; |
| 431 | result[@backingInt(Feature.atomic_fmin_fmax_global_f32)] = .{ |
| 432 | .llvm_name = "atomic-fmin-fmax-global-f32", |
| 433 | .description = "Has global/buffer instructions for atomicrmw fmin/fmax for float", |
| 434 | .dependencies = featureSet(&[_]Feature{}), |
| 435 | }; |
| 436 | result[@backingInt(Feature.atomic_fmin_fmax_global_f64)] = .{ |
| 437 | .llvm_name = "atomic-fmin-fmax-global-f64", |
| 438 | .description = "Has global/buffer instructions for atomicrmw fmin/fmax for float", |
| 439 | .dependencies = featureSet(&[_]Feature{}), |
| 440 | }; |
| 441 | result[@backingInt(Feature.atomic_global_pk_add_bf16_inst)] = .{ |
| 442 | .llvm_name = "atomic-global-pk-add-bf16-inst", |
| 443 | .description = "Has global_atomic_pk_add_bf16 instruction", |
| 444 | .dependencies = featureSet(&[_]Feature{ |
| 445 | .flat_global_insts, |
| 446 | }), |
| 447 | }; |
| 448 | result[@backingInt(Feature.auto_waitcnt_before_barrier)] = .{ |
| 449 | .llvm_name = "auto-waitcnt-before-barrier", |
| 450 | .description = "Hardware automatically inserts waitcnt before barrier", |
| 451 | .dependencies = featureSet(&[_]Feature{}), |
| 452 | }; |
| 453 | result[@backingInt(Feature.back_off_barrier)] = .{ |
| 454 | .llvm_name = "back-off-barrier", |
| 455 | .description = "Hardware supports backing off s_barrier if an exception occurs", |
| 456 | .dependencies = featureSet(&[_]Feature{}), |
| 457 | }; |
| 458 | result[@backingInt(Feature.bf16_cvt_insts)] = .{ |
| 459 | .llvm_name = "bf16-cvt-insts", |
| 460 | .description = "Has bf16 conversion instructions", |
| 461 | .dependencies = featureSet(&[_]Feature{}), |
| 462 | }; |
| 463 | result[@backingInt(Feature.bf16_pk_insts)] = .{ |
| 464 | .llvm_name = "bf16-pk-insts", |
| 465 | .description = "Has bf16 packed instructions (fma, add, mul, max, min)", |
| 466 | .dependencies = featureSet(&[_]Feature{}), |
| 467 | }; |
| 468 | result[@backingInt(Feature.bf16_trans_insts)] = .{ |
| 469 | .llvm_name = "bf16-trans-insts", |
| 470 | .description = "Has bf16 transcendental instructions", |
| 471 | .dependencies = featureSet(&[_]Feature{}), |
| 472 | }; |
| 473 | result[@backingInt(Feature.bf8_cvt_scale_insts)] = .{ |
| 474 | .llvm_name = "bf8-cvt-scale-insts", |
| 475 | .description = "Has bf8 conversion scale instructions", |
| 476 | .dependencies = featureSet(&[_]Feature{}), |
| 477 | }; |
| 478 | result[@backingInt(Feature.bitop3_insts)] = .{ |
| 479 | .llvm_name = "bitop3-insts", |
| 480 | .description = "Has v_bitop3_b32/v_bitop3_b16 instructions", |
| 481 | .dependencies = featureSet(&[_]Feature{}), |
| 482 | }; |
| 483 | result[@backingInt(Feature.block_vgpr_csr)] = .{ |
| 484 | .llvm_name = "block-vgpr-csr", |
| 485 | .description = "Use block load/store for VGPR callee saved registers", |
| 486 | .dependencies = featureSet(&[_]Feature{}), |
| 487 | }; |
| 488 | result[@backingInt(Feature.bvh_dual_bvh_8_insts)] = .{ |
| 489 | .llvm_name = "bvh-dual-bvh-8-insts", |
| 490 | .description = "Has image_bvh_dual_intersect_ray and image_bvh8_intersect_ray instructions", |
| 491 | .dependencies = featureSet(&[_]Feature{}), |
| 492 | }; |
| 493 | result[@backingInt(Feature.ci_insts)] = .{ |
| 494 | .llvm_name = "ci-insts", |
| 495 | .description = "Additional instructions for CI+", |
| 496 | .dependencies = featureSet(&[_]Feature{}), |
| 497 | }; |
| 498 | result[@backingInt(Feature.clusters)] = .{ |
| 499 | .llvm_name = "clusters", |
| 500 | .description = "Has clusters of workgroups support", |
| 501 | .dependencies = featureSet(&[_]Feature{}), |
| 502 | }; |
| 503 | result[@backingInt(Feature.cube_insts)] = .{ |
| 504 | .llvm_name = "cube-insts", |
| 505 | .description = "Has v_cube* instructions", |
| 506 | .dependencies = featureSet(&[_]Feature{}), |
| 507 | }; |
| 508 | result[@backingInt(Feature.cumode)] = .{ |
| 509 | .llvm_name = "cumode", |
| 510 | .description = "Enable CU wavefront execution mode", |
| 511 | .dependencies = featureSet(&[_]Feature{}), |
| 512 | }; |
| 513 | result[@backingInt(Feature.cvt_fp8_vop1_bug)] = .{ |
| 514 | .llvm_name = "cvt-fp8-vop1-bug", |
| 515 | .description = "FP8/BF8 VOP1 form of conversion to F32 is unreliable", |
| 516 | .dependencies = featureSet(&[_]Feature{ |
| 517 | .fp8_conversion_insts, |
| 518 | }), |
| 519 | }; |
| 520 | result[@backingInt(Feature.cvt_norm_insts)] = .{ |
| 521 | .llvm_name = "cvt-norm-insts", |
| 522 | .description = "Has v_cvt_norm* instructions", |
| 523 | .dependencies = featureSet(&[_]Feature{}), |
| 524 | }; |
| 525 | result[@backingInt(Feature.cvt_pk_f16_f32_inst)] = .{ |
| 526 | .llvm_name = "cvt-pk-f16-f32-inst", |
| 527 | .description = "Has cvt_pk_f16_f32 instruction", |
| 528 | .dependencies = featureSet(&[_]Feature{}), |
| 529 | }; |
| 530 | result[@backingInt(Feature.cvt_pknorm_vop2_insts)] = .{ |
| 531 | .llvm_name = "cvt-pknorm-vop2-insts", |
| 532 | .description = "Has v_cvt_pk_norm_*f32 instructions/Has v_cvt_pk_norm_*_f16 instructions", |
| 533 | .dependencies = featureSet(&[_]Feature{}), |
| 534 | }; |
| 535 | result[@backingInt(Feature.cvt_pknorm_vop3_insts)] = .{ |
| 536 | .llvm_name = "cvt-pknorm-vop3-insts", |
| 537 | .description = "Has v_cvt_pk_norm_*f32 instructions/Has v_cvt_pk_norm_*_f16 instructions", |
| 538 | .dependencies = featureSet(&[_]Feature{}), |
| 539 | }; |
| 540 | result[@backingInt(Feature.d16_write_vgpr32)] = .{ |
| 541 | .llvm_name = "d16-write-vgpr32", |
| 542 | .description = "D16 instructions potentially have 32-bit data dependencies", |
| 543 | .dependencies = featureSet(&[_]Feature{}), |
| 544 | }; |
| 545 | result[@backingInt(Feature.default_component_broadcast)] = .{ |
| 546 | .llvm_name = "default-component-broadcast", |
| 547 | .description = "BUFFER/IMAGE store instructions set unspecified components to x component (GFX12)", |
| 548 | .dependencies = featureSet(&[_]Feature{}), |
| 549 | }; |
| 550 | result[@backingInt(Feature.default_component_zero)] = .{ |
| 551 | .llvm_name = "default-component-zero", |
| 552 | .description = "BUFFER/IMAGE store instructions set unspecified components to zero (before GFX12)", |
| 553 | .dependencies = featureSet(&[_]Feature{}), |
| 554 | }; |
| 555 | result[@backingInt(Feature.dl_insts)] = .{ |
| 556 | .llvm_name = "dl-insts", |
| 557 | .description = "Has v_fmac_f32 and v_xnor_b32 instructions", |
| 558 | .dependencies = featureSet(&[_]Feature{}), |
| 559 | }; |
| 560 | result[@backingInt(Feature.dot10_insts)] = .{ |
| 561 | .llvm_name = "dot10-insts", |
| 562 | .description = "Has v_dot2_f32_f16 instruction", |
| 563 | .dependencies = featureSet(&[_]Feature{}), |
| 564 | }; |
| 565 | result[@backingInt(Feature.dot11_insts)] = .{ |
| 566 | .llvm_name = "dot11-insts", |
| 567 | .description = "Has v_dot4_f32_fp8_fp8, v_dot4_f32_fp8_bf8, v_dot4_f32_bf8_fp8, v_dot4_f32_bf8_bf8 instructions", |
| 568 | .dependencies = featureSet(&[_]Feature{}), |
| 569 | }; |
| 570 | result[@backingInt(Feature.dot12_insts)] = .{ |
| 571 | .llvm_name = "dot12-insts", |
| 572 | .description = "Has v_dot2_f32_bf16 instructions", |
| 573 | .dependencies = featureSet(&[_]Feature{}), |
| 574 | }; |
| 575 | result[@backingInt(Feature.dot13_insts)] = .{ |
| 576 | .llvm_name = "dot13-insts", |
| 577 | .description = "Has v_dot2c_f32_bf16 instructions", |
| 578 | .dependencies = featureSet(&[_]Feature{}), |
| 579 | }; |
| 580 | result[@backingInt(Feature.dot1_insts)] = .{ |
| 581 | .llvm_name = "dot1-insts", |
| 582 | .description = "Has v_dot4_i32_i8 and v_dot8_i32_i4 instructions", |
| 583 | .dependencies = featureSet(&[_]Feature{}), |
| 584 | }; |
| 585 | result[@backingInt(Feature.dot2_insts)] = .{ |
| 586 | .llvm_name = "dot2-insts", |
| 587 | .description = "Has v_dot2_i32_i16, v_dot2_u32_u16 instructions", |
| 588 | .dependencies = featureSet(&[_]Feature{}), |
| 589 | }; |
| 590 | result[@backingInt(Feature.dot3_insts)] = .{ |
| 591 | .llvm_name = "dot3-insts", |
| 592 | .description = "Has v_dot8c_i32_i4 instruction", |
| 593 | .dependencies = featureSet(&[_]Feature{}), |
| 594 | }; |
| 595 | result[@backingInt(Feature.dot4_insts)] = .{ |
| 596 | .llvm_name = "dot4-insts", |
| 597 | .description = "Has v_dot2c_i32_i16 instruction", |
| 598 | .dependencies = featureSet(&[_]Feature{}), |
| 599 | }; |
| 600 | result[@backingInt(Feature.dot5_insts)] = .{ |
| 601 | .llvm_name = "dot5-insts", |
| 602 | .description = "Has v_dot2c_f32_f16 instruction", |
| 603 | .dependencies = featureSet(&[_]Feature{}), |
| 604 | }; |
| 605 | result[@backingInt(Feature.dot6_insts)] = .{ |
| 606 | .llvm_name = "dot6-insts", |
| 607 | .description = "Has v_dot4c_i32_i8 instruction", |
| 608 | .dependencies = featureSet(&[_]Feature{}), |
| 609 | }; |
| 610 | result[@backingInt(Feature.dot7_insts)] = .{ |
| 611 | .llvm_name = "dot7-insts", |
| 612 | .description = "Has v_dot4_u32_u8, v_dot8_u32_u4 instructions", |
| 613 | .dependencies = featureSet(&[_]Feature{}), |
| 614 | }; |
| 615 | result[@backingInt(Feature.dot8_insts)] = .{ |
| 616 | .llvm_name = "dot8-insts", |
| 617 | .description = "Has v_dot4_i32_iu8, v_dot8_i32_iu4 instructions", |
| 618 | .dependencies = featureSet(&[_]Feature{}), |
| 619 | }; |
| 620 | result[@backingInt(Feature.dot9_insts)] = .{ |
| 621 | .llvm_name = "dot9-insts", |
| 622 | .description = "Has v_dot2_f16_f16, v_dot2_bf16_bf16 instructions", |
| 623 | .dependencies = featureSet(&[_]Feature{}), |
| 624 | }; |
| 625 | result[@backingInt(Feature.dpp)] = .{ |
| 626 | .llvm_name = "dpp", |
| 627 | .description = "Support DPP (Data Parallel Primitives) extension", |
| 628 | .dependencies = featureSet(&[_]Feature{}), |
| 629 | }; |
| 630 | result[@backingInt(Feature.dpp8)] = .{ |
| 631 | .llvm_name = "dpp8", |
| 632 | .description = "Support DPP8 (Data Parallel Primitives) extension", |
| 633 | .dependencies = featureSet(&[_]Feature{}), |
| 634 | }; |
| 635 | result[@backingInt(Feature.dpp_64bit)] = .{ |
| 636 | .llvm_name = "dpp-64bit", |
| 637 | .description = "Support DPP (Data Parallel Primitives) extension in DP ALU", |
| 638 | .dependencies = featureSet(&[_]Feature{}), |
| 639 | }; |
| 640 | result[@backingInt(Feature.dpp_src1_sgpr)] = .{ |
| 641 | .llvm_name = "dpp-src1-sgpr", |
| 642 | .description = "Support SGPR for Src1 of DPP instructions", |
| 643 | .dependencies = featureSet(&[_]Feature{}), |
| 644 | }; |
| 645 | result[@backingInt(Feature.ds128)] = .{ |
| 646 | .llvm_name = "enable-ds128", |
| 647 | .description = "Use ds_{read|write}_b128", |
| 648 | .dependencies = featureSet(&[_]Feature{}), |
| 649 | }; |
| 650 | result[@backingInt(Feature.ds_src2_insts)] = .{ |
| 651 | .llvm_name = "ds-src2-insts", |
| 652 | .description = "Has ds_*_src2 instructions", |
| 653 | .dependencies = featureSet(&[_]Feature{}), |
| 654 | }; |
| 655 | result[@backingInt(Feature.emulated_system_scope_atomics)] = .{ |
| 656 | .llvm_name = "emulated-system-scope-atomics", |
| 657 | .description = "System scope atomics unsupported by the PCI-e are emulated in HW via CAS loop and functional.", |
| 658 | .dependencies = featureSet(&[_]Feature{}), |
| 659 | }; |
| 660 | result[@backingInt(Feature.extended_image_insts)] = .{ |
| 661 | .llvm_name = "extended-image-insts", |
| 662 | .description = "Support mips != 0, lod != 0, gather4, and get_lod", |
| 663 | .dependencies = featureSet(&[_]Feature{}), |
| 664 | }; |
| 665 | result[@backingInt(Feature.f16bf16_to_fp6bf6_cvt_scale_insts)] = .{ |
| 666 | .llvm_name = "f16bf16-to-fp6bf6-cvt-scale-insts", |
| 667 | .description = "Has f16bf16 to fp6bf6 conversion scale instructions", |
| 668 | .dependencies = featureSet(&[_]Feature{}), |
| 669 | }; |
| 670 | result[@backingInt(Feature.f32_to_f16bf16_cvt_sr_insts)] = .{ |
| 671 | .llvm_name = "f32-to-f16bf16-cvt-sr-insts", |
| 672 | .description = "Has f32 to f16bf16 conversion scale instructions", |
| 673 | .dependencies = featureSet(&[_]Feature{}), |
| 674 | }; |
| 675 | result[@backingInt(Feature.fast_denormal_f32)] = .{ |
| 676 | .llvm_name = "fast-denormal-f32", |
| 677 | .description = "Enabling denormals does not cause f32 instructions to run at f64 rates", |
| 678 | .dependencies = featureSet(&[_]Feature{}), |
| 679 | }; |
| 680 | result[@backingInt(Feature.fast_fmaf)] = .{ |
| 681 | .llvm_name = "fast-fmaf", |
| 682 | .description = "Assuming f32 fma is at least as fast as mul + add", |
| 683 | .dependencies = featureSet(&[_]Feature{}), |
| 684 | }; |
| 685 | result[@backingInt(Feature.flat_address_space)] = .{ |
| 686 | .llvm_name = "flat-address-space", |
| 687 | .description = "Support flat address space", |
| 688 | .dependencies = featureSet(&[_]Feature{}), |
| 689 | }; |
| 690 | result[@backingInt(Feature.flat_atomic_fadd_f32_inst)] = .{ |
| 691 | .llvm_name = "flat-atomic-fadd-f32-inst", |
| 692 | .description = "Has flat_atomic_add_f32 instruction", |
| 693 | .dependencies = featureSet(&[_]Feature{ |
| 694 | .flat_address_space, |
| 695 | }), |
| 696 | }; |
| 697 | result[@backingInt(Feature.flat_buffer_global_fadd_f64_inst)] = .{ |
| 698 | .llvm_name = "flat-buffer-global-fadd-f64-inst", |
| 699 | .description = "Has flat, buffer, and global instructions for f64 atomic fadd", |
| 700 | .dependencies = featureSet(&[_]Feature{}), |
| 701 | }; |
| 702 | result[@backingInt(Feature.flat_for_global)] = .{ |
| 703 | .llvm_name = "flat-for-global", |
| 704 | .description = "Force to generate flat instruction for global", |
| 705 | .dependencies = featureSet(&[_]Feature{}), |
| 706 | }; |
| 707 | result[@backingInt(Feature.flat_global_insts)] = .{ |
| 708 | .llvm_name = "flat-global-insts", |
| 709 | .description = "Have global_* flat memory instructions", |
| 710 | .dependencies = featureSet(&[_]Feature{ |
| 711 | .flat_address_space, |
| 712 | }), |
| 713 | }; |
| 714 | result[@backingInt(Feature.flat_gvs_mode)] = .{ |
| 715 | .llvm_name = "flat-gvs-mode", |
| 716 | .description = "Have GVS addressing mode with flat_* instructions", |
| 717 | .dependencies = featureSet(&[_]Feature{ |
| 718 | .flat_address_space, |
| 719 | }), |
| 720 | }; |
| 721 | result[@backingInt(Feature.flat_inst_offsets)] = .{ |
| 722 | .llvm_name = "flat-inst-offsets", |
| 723 | .description = "Flat instructions have immediate offset addressing mode", |
| 724 | .dependencies = featureSet(&[_]Feature{}), |
| 725 | }; |
| 726 | result[@backingInt(Feature.flat_scratch)] = .{ |
| 727 | .llvm_name = "enable-flat-scratch", |
| 728 | .description = "Use scratch_* flat memory instructions to access scratch", |
| 729 | .dependencies = featureSet(&[_]Feature{}), |
| 730 | }; |
| 731 | result[@backingInt(Feature.flat_scratch_insts)] = .{ |
| 732 | .llvm_name = "flat-scratch-insts", |
| 733 | .description = "Have scratch_* flat memory instructions", |
| 734 | .dependencies = featureSet(&[_]Feature{ |
| 735 | .flat_address_space, |
| 736 | }), |
| 737 | }; |
| 738 | result[@backingInt(Feature.flat_segment_offset_bug)] = .{ |
| 739 | .llvm_name = "flat-segment-offset-bug", |
| 740 | .description = "GFX10 bug where inst_offset is ignored when flat instructions access global memory", |
| 741 | .dependencies = featureSet(&[_]Feature{}), |
| 742 | }; |
| 743 | result[@backingInt(Feature.fma_mix_bf16_insts)] = .{ |
| 744 | .llvm_name = "fma-mix-bf16-insts", |
| 745 | .description = "Has v_fma_mix_f32_bf16, v_fma_mixlo_bf16, v_fma_mixhi_bf16 instructions", |
| 746 | .dependencies = featureSet(&[_]Feature{}), |
| 747 | }; |
| 748 | result[@backingInt(Feature.fma_mix_insts)] = .{ |
| 749 | .llvm_name = "fma-mix-insts", |
| 750 | .description = "Has v_fma_mix_f32, v_fma_mixlo_f16, v_fma_mixhi_f16 instructions", |
| 751 | .dependencies = featureSet(&[_]Feature{}), |
| 752 | }; |
| 753 | result[@backingInt(Feature.fmacf64_inst)] = .{ |
| 754 | .llvm_name = "fmacf64-inst", |
| 755 | .description = "Has v_fmac_f64 instruction", |
| 756 | .dependencies = featureSet(&[_]Feature{}), |
| 757 | }; |
| 758 | result[@backingInt(Feature.fmaf)] = .{ |
| 759 | .llvm_name = "fmaf", |
| 760 | .description = "Enable single precision FMA (not as fast as mul+add, but fused)", |
| 761 | .dependencies = featureSet(&[_]Feature{}), |
| 762 | }; |
| 763 | result[@backingInt(Feature.fp4_cvt_scale_insts)] = .{ |
| 764 | .llvm_name = "fp4-cvt-scale-insts", |
| 765 | .description = "Has fp4 conversion scale instructions", |
| 766 | .dependencies = featureSet(&[_]Feature{}), |
| 767 | }; |
| 768 | result[@backingInt(Feature.fp64)] = .{ |
| 769 | .llvm_name = "fp64", |
| 770 | .description = "Enable double precision operations", |
| 771 | .dependencies = featureSet(&[_]Feature{}), |
| 772 | }; |
| 773 | result[@backingInt(Feature.fp6bf6_cvt_scale_insts)] = .{ |
| 774 | .llvm_name = "fp6bf6-cvt-scale-insts", |
| 775 | .description = "Has fp6 and bf6 conversion scale instructions", |
| 776 | .dependencies = featureSet(&[_]Feature{}), |
| 777 | }; |
| 778 | result[@backingInt(Feature.fp8_conversion_insts)] = .{ |
| 779 | .llvm_name = "fp8-conversion-insts", |
| 780 | .description = "Has fp8 and bf8 conversion instructions", |
| 781 | .dependencies = featureSet(&[_]Feature{}), |
| 782 | }; |
| 783 | result[@backingInt(Feature.fp8_cvt_scale_insts)] = .{ |
| 784 | .llvm_name = "fp8-cvt-scale-insts", |
| 785 | .description = "Has fp8 conversion scale instructions", |
| 786 | .dependencies = featureSet(&[_]Feature{}), |
| 787 | }; |
| 788 | result[@backingInt(Feature.fp8_insts)] = .{ |
| 789 | .llvm_name = "fp8-insts", |
| 790 | .description = "Has fp8 and bf8 instructions", |
| 791 | .dependencies = featureSet(&[_]Feature{}), |
| 792 | }; |
| 793 | result[@backingInt(Feature.fp8e5m3_insts)] = .{ |
| 794 | .llvm_name = "fp8e5m3-insts", |
| 795 | .description = "Has fp8 e5m3 format support", |
| 796 | .dependencies = featureSet(&[_]Feature{}), |
| 797 | }; |
| 798 | result[@backingInt(Feature.full_rate_64_ops)] = .{ |
| 799 | .llvm_name = "full-rate-64-ops", |
| 800 | .description = "Most fp64 instructions are full rate", |
| 801 | .dependencies = featureSet(&[_]Feature{}), |
| 802 | }; |
| 803 | result[@backingInt(Feature.g16)] = .{ |
| 804 | .llvm_name = "g16", |
| 805 | .description = "Support G16 for 16-bit gradient image operands", |
| 806 | .dependencies = featureSet(&[_]Feature{}), |
| 807 | }; |
| 808 | result[@backingInt(Feature.gcn3_encoding)] = .{ |
| 809 | .llvm_name = "gcn3-encoding", |
| 810 | .description = "Encoding format for VI", |
| 811 | .dependencies = featureSet(&[_]Feature{}), |
| 812 | }; |
| 813 | result[@backingInt(Feature.gds)] = .{ |
| 814 | .llvm_name = "gds", |
| 815 | .description = "Has Global Data Share", |
| 816 | .dependencies = featureSet(&[_]Feature{}), |
| 817 | }; |
| 818 | result[@backingInt(Feature.get_wave_id_inst)] = .{ |
| 819 | .llvm_name = "get-wave-id-inst", |
| 820 | .description = "Has s_get_waveid_in_workgroup instruction", |
| 821 | .dependencies = featureSet(&[_]Feature{}), |
| 822 | }; |
| 823 | result[@backingInt(Feature.gfx10)] = .{ |
| 824 | .llvm_name = "gfx10", |
| 825 | .description = "GFX10 GPU generation", |
| 826 | .dependencies = featureSet(&[_]Feature{ |
| 827 | .@"16_bit_insts", |
| 828 | .a16, |
| 829 | .add_no_carry_insts, |
| 830 | .addressablelocalmemorysize65536, |
| 831 | .aperture_regs, |
| 832 | .atomic_fmin_fmax_flat_f32, |
| 833 | .atomic_fmin_fmax_flat_f64, |
| 834 | .atomic_fmin_fmax_global_f32, |
| 835 | .atomic_fmin_fmax_global_f64, |
| 836 | .ci_insts, |
| 837 | .cube_insts, |
| 838 | .cvt_norm_insts, |
| 839 | .cvt_pknorm_vop2_insts, |
| 840 | .cvt_pknorm_vop3_insts, |
| 841 | .default_component_zero, |
| 842 | .dpp, |
| 843 | .dpp8, |
| 844 | .extended_image_insts, |
| 845 | .fast_denormal_f32, |
| 846 | .fast_fmaf, |
| 847 | .flat_global_insts, |
| 848 | .flat_inst_offsets, |
| 849 | .flat_scratch_insts, |
| 850 | .fma_mix_insts, |
| 851 | .fp64, |
| 852 | .g16, |
| 853 | .gds, |
| 854 | .gfx10_insts, |
| 855 | .gfx8_insts, |
| 856 | .gfx9_insts, |
| 857 | .gws, |
| 858 | .image_insts, |
| 859 | .int_clamp_insts, |
| 860 | .inv_2pi_inline_imm, |
| 861 | .lerp_inst, |
| 862 | .max_hard_clause_length_63, |
| 863 | .mimg_r128, |
| 864 | .movrel, |
| 865 | .no_data_dep_hazard, |
| 866 | .no_sdst_cmpx, |
| 867 | .pk_fmac_f16_inst, |
| 868 | .qsad_insts, |
| 869 | .s_memrealtime, |
| 870 | .s_memtime_inst, |
| 871 | .sad_insts, |
| 872 | .sdwa, |
| 873 | .sdwa_omod, |
| 874 | .sdwa_scalar, |
| 875 | .sdwa_sdst, |
| 876 | .unaligned_buffer_access, |
| 877 | .unaligned_ds_access, |
| 878 | .unaligned_scratch_access, |
| 879 | .vmem_to_lds_load_insts, |
| 880 | .vmem_write_vgpr_in_order, |
| 881 | .vop3_literal, |
| 882 | .vop3p, |
| 883 | .vscnt, |
| 884 | }), |
| 885 | }; |
| 886 | result[@backingInt(Feature.gfx10_3_insts)] = .{ |
| 887 | .llvm_name = "gfx10-3-insts", |
| 888 | .description = "Additional instructions for GFX10.3", |
| 889 | .dependencies = featureSet(&[_]Feature{}), |
| 890 | }; |
| 891 | result[@backingInt(Feature.gfx10_a_encoding)] = .{ |
| 892 | .llvm_name = "gfx10_a-encoding", |
| 893 | .description = "Has BVH ray tracing instructions", |
| 894 | .dependencies = featureSet(&[_]Feature{}), |
| 895 | }; |
| 896 | result[@backingInt(Feature.gfx10_b_encoding)] = .{ |
| 897 | .llvm_name = "gfx10_b-encoding", |
| 898 | .description = "Encoding format GFX10_B", |
| 899 | .dependencies = featureSet(&[_]Feature{}), |
| 900 | }; |
| 901 | result[@backingInt(Feature.gfx10_insts)] = .{ |
| 902 | .llvm_name = "gfx10-insts", |
| 903 | .description = "Additional instructions for GFX10+", |
| 904 | .dependencies = featureSet(&[_]Feature{}), |
| 905 | }; |
| 906 | result[@backingInt(Feature.gfx11)] = .{ |
| 907 | .llvm_name = "gfx11", |
| 908 | .description = "GFX11 GPU generation", |
| 909 | .dependencies = featureSet(&[_]Feature{ |
| 910 | .@"16_bit_insts", |
| 911 | .a16, |
| 912 | .add_no_carry_insts, |
| 913 | .addressablelocalmemorysize65536, |
| 914 | .aperture_regs, |
| 915 | .atomic_fmin_fmax_flat_f32, |
| 916 | .atomic_fmin_fmax_global_f32, |
| 917 | .ci_insts, |
| 918 | .cube_insts, |
| 919 | .cvt_norm_insts, |
| 920 | .cvt_pknorm_vop2_insts, |
| 921 | .cvt_pknorm_vop3_insts, |
| 922 | .default_component_zero, |
| 923 | .dpp, |
| 924 | .dpp8, |
| 925 | .extended_image_insts, |
| 926 | .fast_denormal_f32, |
| 927 | .fast_fmaf, |
| 928 | .flat_global_insts, |
| 929 | .flat_inst_offsets, |
| 930 | .flat_scratch_insts, |
| 931 | .fma_mix_insts, |
| 932 | .fp64, |
| 933 | .g16, |
| 934 | .gds, |
| 935 | .gfx10_3_insts, |
| 936 | .gfx10_a_encoding, |
| 937 | .gfx10_b_encoding, |
| 938 | .gfx10_insts, |
| 939 | .gfx11_insts, |
| 940 | .gfx8_insts, |
| 941 | .gfx9_insts, |
| 942 | .gws, |
| 943 | .int_clamp_insts, |
| 944 | .inv_2pi_inline_imm, |
| 945 | .lerp_inst, |
| 946 | .max_hard_clause_length_32, |
| 947 | .mimg_r128, |
| 948 | .movrel, |
| 949 | .no_data_dep_hazard, |
| 950 | .no_sdst_cmpx, |
| 951 | .pk_fmac_f16_inst, |
| 952 | .qsad_insts, |
| 953 | .sad_insts, |
| 954 | .true16, |
| 955 | .unaligned_buffer_access, |
| 956 | .unaligned_ds_access, |
| 957 | .unaligned_scratch_access, |
| 958 | .vmem_write_vgpr_in_order, |
| 959 | .vop3_literal, |
| 960 | .vop3p, |
| 961 | .vopd, |
| 962 | .vscnt, |
| 963 | }), |
| 964 | }; |
| 965 | result[@backingInt(Feature.gfx11_insts)] = .{ |
| 966 | .llvm_name = "gfx11-insts", |
| 967 | .description = "Additional instructions for GFX11+", |
| 968 | .dependencies = featureSet(&[_]Feature{}), |
| 969 | }; |
| 970 | result[@backingInt(Feature.gfx12)] = .{ |
| 971 | .llvm_name = "gfx12", |
| 972 | .description = "GFX12 GPU generation", |
| 973 | .dependencies = featureSet(&[_]Feature{ |
| 974 | .@"16_bit_insts", |
| 975 | .a16, |
| 976 | .add_no_carry_insts, |
| 977 | .agent_scope_fine_grained_remote_memory_atomics, |
| 978 | .aperture_regs, |
| 979 | .atomic_fmin_fmax_flat_f32, |
| 980 | .atomic_fmin_fmax_global_f32, |
| 981 | .ci_insts, |
| 982 | .default_component_broadcast, |
| 983 | .dpp, |
| 984 | .dpp8, |
| 985 | .fast_denormal_f32, |
| 986 | .fast_fmaf, |
| 987 | .flat_global_insts, |
| 988 | .flat_inst_offsets, |
| 989 | .flat_scratch_insts, |
| 990 | .fma_mix_insts, |
| 991 | .fp64, |
| 992 | .g16, |
| 993 | .gfx10_3_insts, |
| 994 | .gfx10_a_encoding, |
| 995 | .gfx10_b_encoding, |
| 996 | .gfx10_insts, |
| 997 | .gfx11_insts, |
| 998 | .gfx12_insts, |
| 999 | .gfx8_insts, |
| 1000 | .gfx9_insts, |
| 1001 | .ieee_minimum_maximum_insts, |
| 1002 | .int_clamp_insts, |
| 1003 | .inv_2pi_inline_imm, |
| 1004 | .max_hard_clause_length_32, |
| 1005 | .mimg_r128, |
| 1006 | .minimum3_maximum3_f16, |
| 1007 | .minimum3_maximum3_f32, |
| 1008 | .movrel, |
| 1009 | .no_data_dep_hazard, |
| 1010 | .no_sdst_cmpx, |
| 1011 | .pk_fmac_f16_inst, |
| 1012 | .true16, |
| 1013 | .unaligned_buffer_access, |
| 1014 | .unaligned_ds_access, |
| 1015 | .unaligned_scratch_access, |
| 1016 | .vop3_literal, |
| 1017 | .vop3p, |
| 1018 | .vopd, |
| 1019 | .vscnt, |
| 1020 | }), |
| 1021 | }; |
| 1022 | result[@backingInt(Feature.gfx1250_insts)] = .{ |
| 1023 | .llvm_name = "gfx1250-insts", |
| 1024 | .description = "Additional instructions for GFX1250+", |
| 1025 | .dependencies = featureSet(&[_]Feature{}), |
| 1026 | }; |
| 1027 | result[@backingInt(Feature.gfx12_insts)] = .{ |
| 1028 | .llvm_name = "gfx12-insts", |
| 1029 | .description = "Additional instructions for GFX12+", |
| 1030 | .dependencies = featureSet(&[_]Feature{}), |
| 1031 | }; |
| 1032 | result[@backingInt(Feature.gfx7_gfx8_gfx9_insts)] = .{ |
| 1033 | .llvm_name = "gfx7-gfx8-gfx9-insts", |
| 1034 | .description = "Instructions shared in GFX7, GFX8, GFX9", |
| 1035 | .dependencies = featureSet(&[_]Feature{}), |
| 1036 | }; |
| 1037 | result[@backingInt(Feature.gfx8_insts)] = .{ |
| 1038 | .llvm_name = "gfx8-insts", |
| 1039 | .description = "Additional instructions for GFX8+", |
| 1040 | .dependencies = featureSet(&[_]Feature{}), |
| 1041 | }; |
| 1042 | result[@backingInt(Feature.gfx9)] = .{ |
| 1043 | .llvm_name = "gfx9", |
| 1044 | .description = "GFX9 GPU generation", |
| 1045 | .dependencies = featureSet(&[_]Feature{ |
| 1046 | .@"16_bit_insts", |
| 1047 | .a16, |
| 1048 | .add_no_carry_insts, |
| 1049 | .aperture_regs, |
| 1050 | .ci_insts, |
| 1051 | .cube_insts, |
| 1052 | .cvt_norm_insts, |
| 1053 | .cvt_pknorm_vop2_insts, |
| 1054 | .cvt_pknorm_vop3_insts, |
| 1055 | .default_component_zero, |
| 1056 | .dpp, |
| 1057 | .fast_denormal_f32, |
| 1058 | .fast_fmaf, |
| 1059 | .flat_global_insts, |
| 1060 | .flat_inst_offsets, |
| 1061 | .flat_scratch_insts, |
| 1062 | .fp64, |
| 1063 | .gcn3_encoding, |
| 1064 | .gfx7_gfx8_gfx9_insts, |
| 1065 | .gfx8_insts, |
| 1066 | .gfx9_insts, |
| 1067 | .gws, |
| 1068 | .int_clamp_insts, |
| 1069 | .inv_2pi_inline_imm, |
| 1070 | .lerp_inst, |
| 1071 | .negative_scratch_offset_bug, |
| 1072 | .qsad_insts, |
| 1073 | .r128_a16, |
| 1074 | .s_memrealtime, |
| 1075 | .s_memtime_inst, |
| 1076 | .sad_insts, |
| 1077 | .scalar_atomics, |
| 1078 | .scalar_flat_scratch_insts, |
| 1079 | .scalar_stores, |
| 1080 | .sdwa, |
| 1081 | .sdwa_omod, |
| 1082 | .sdwa_scalar, |
| 1083 | .sdwa_sdst, |
| 1084 | .unaligned_buffer_access, |
| 1085 | .unaligned_ds_access, |
| 1086 | .unaligned_scratch_access, |
| 1087 | .vgpr_index_mode, |
| 1088 | .vmem_to_lds_load_insts, |
| 1089 | .vmem_write_vgpr_in_order, |
| 1090 | .vop3p, |
| 1091 | .wavefrontsize64, |
| 1092 | .xnack_support, |
| 1093 | }), |
| 1094 | }; |
| 1095 | result[@backingInt(Feature.gfx90a_insts)] = .{ |
| 1096 | .llvm_name = "gfx90a-insts", |
| 1097 | .description = "Additional instructions for GFX90A+", |
| 1098 | .dependencies = featureSet(&[_]Feature{}), |
| 1099 | }; |
| 1100 | result[@backingInt(Feature.gfx940_insts)] = .{ |
| 1101 | .llvm_name = "gfx940-insts", |
| 1102 | .description = "Additional instructions for GFX940+", |
| 1103 | .dependencies = featureSet(&[_]Feature{}), |
| 1104 | }; |
| 1105 | result[@backingInt(Feature.gfx950_insts)] = .{ |
| 1106 | .llvm_name = "gfx950-insts", |
| 1107 | .description = "Additional instructions for GFX950+", |
| 1108 | .dependencies = featureSet(&[_]Feature{ |
| 1109 | .ashr_pk_insts, |
| 1110 | .bf8_cvt_scale_insts, |
| 1111 | .cvt_pk_f16_f32_inst, |
| 1112 | .f16bf16_to_fp6bf6_cvt_scale_insts, |
| 1113 | .f32_to_f16bf16_cvt_sr_insts, |
| 1114 | .fp4_cvt_scale_insts, |
| 1115 | .fp6bf6_cvt_scale_insts, |
| 1116 | .fp8_cvt_scale_insts, |
| 1117 | .minimum3_maximum3_f32, |
| 1118 | .minimum3_maximum3_pkf16, |
| 1119 | .permlane16_swap, |
| 1120 | .permlane32_swap, |
| 1121 | }), |
| 1122 | }; |
| 1123 | result[@backingInt(Feature.gfx9_insts)] = .{ |
| 1124 | .llvm_name = "gfx9-insts", |
| 1125 | .description = "Additional instructions for GFX9+", |
| 1126 | .dependencies = featureSet(&[_]Feature{}), |
| 1127 | }; |
| 1128 | result[@backingInt(Feature.globally_addressable_scratch)] = .{ |
| 1129 | .llvm_name = "globally-addressable-scratch", |
| 1130 | .description = "FLAT instructions can access scratch memory for any thread in any wave", |
| 1131 | .dependencies = featureSet(&[_]Feature{}), |
| 1132 | }; |
| 1133 | result[@backingInt(Feature.gws)] = .{ |
| 1134 | .llvm_name = "gws", |
| 1135 | .description = "Has Global Wave Sync", |
| 1136 | .dependencies = featureSet(&[_]Feature{}), |
| 1137 | }; |
| 1138 | result[@backingInt(Feature.half_rate_64_ops)] = .{ |
| 1139 | .llvm_name = "half-rate-64-ops", |
| 1140 | .description = "Most fp64 instructions are half rate instead of quarter", |
| 1141 | .dependencies = featureSet(&[_]Feature{}), |
| 1142 | }; |
| 1143 | result[@backingInt(Feature.ieee_minimum_maximum_insts)] = .{ |
| 1144 | .llvm_name = "ieee-minimum-maximum-insts", |
| 1145 | .description = "Has v_minimum/maximum_f16/f32/f64, v_minimummaximum/maximumminimum_f16/f32 and v_pk_minimum/maximum_f16 instructions", |
| 1146 | .dependencies = featureSet(&[_]Feature{}), |
| 1147 | }; |
| 1148 | result[@backingInt(Feature.image_gather4_d16_bug)] = .{ |
| 1149 | .llvm_name = "image-gather4-d16-bug", |
| 1150 | .description = "Image Gather4 D16 hardware bug", |
| 1151 | .dependencies = featureSet(&[_]Feature{}), |
| 1152 | }; |
| 1153 | result[@backingInt(Feature.image_insts)] = .{ |
| 1154 | .llvm_name = "image-insts", |
| 1155 | .description = "Support image instructions", |
| 1156 | .dependencies = featureSet(&[_]Feature{}), |
| 1157 | }; |
| 1158 | result[@backingInt(Feature.image_store_d16_bug)] = .{ |
| 1159 | .llvm_name = "image-store-d16-bug", |
| 1160 | .description = "Image Store D16 hardware bug", |
| 1161 | .dependencies = featureSet(&[_]Feature{}), |
| 1162 | }; |
| 1163 | result[@backingInt(Feature.inst_fwd_prefetch_bug)] = .{ |
| 1164 | .llvm_name = "inst-fwd-prefetch-bug", |
| 1165 | .description = "S_INST_PREFETCH instruction causes shader to hang", |
| 1166 | .dependencies = featureSet(&[_]Feature{}), |
| 1167 | }; |
| 1168 | result[@backingInt(Feature.int_clamp_insts)] = .{ |
| 1169 | .llvm_name = "int-clamp-insts", |
| 1170 | .description = "Support clamp for integer destination", |
| 1171 | .dependencies = featureSet(&[_]Feature{}), |
| 1172 | }; |
| 1173 | result[@backingInt(Feature.inv_2pi_inline_imm)] = .{ |
| 1174 | .llvm_name = "inv-2pi-inline-imm", |
| 1175 | .description = "Has 1 / (2 * pi) as inline immediate", |
| 1176 | .dependencies = featureSet(&[_]Feature{}), |
| 1177 | }; |
| 1178 | result[@backingInt(Feature.kernarg_preload)] = .{ |
| 1179 | .llvm_name = "kernarg-preload", |
| 1180 | .description = "Hardware supports preloading of kernel arguments in user SGPRs.", |
| 1181 | .dependencies = featureSet(&[_]Feature{}), |
| 1182 | }; |
| 1183 | result[@backingInt(Feature.lds_barrier_arrive_atomic)] = .{ |
| 1184 | .llvm_name = "lds-barrier-arrive-atomic", |
| 1185 | .description = "Has LDS barrier-arrive atomic instructions", |
| 1186 | .dependencies = featureSet(&[_]Feature{}), |
| 1187 | }; |
| 1188 | result[@backingInt(Feature.lds_branch_vmem_war_hazard)] = .{ |
| 1189 | .llvm_name = "lds-branch-vmem-war-hazard", |
| 1190 | .description = "Switching between LDS and VMEM-tex not waiting VM_VSRC=0", |
| 1191 | .dependencies = featureSet(&[_]Feature{}), |
| 1192 | }; |
| 1193 | result[@backingInt(Feature.lds_misaligned_bug)] = .{ |
| 1194 | .llvm_name = "lds-misaligned-bug", |
| 1195 | .description = "Some GFX10 bug with multi-dword LDS and flat access that is not naturally aligned in WGP mode", |
| 1196 | .dependencies = featureSet(&[_]Feature{}), |
| 1197 | }; |
| 1198 | result[@backingInt(Feature.ldsbankcount16)] = .{ |
| 1199 | .llvm_name = "ldsbankcount16", |
| 1200 | .description = "The number of LDS banks per compute unit.", |
| 1201 | .dependencies = featureSet(&[_]Feature{}), |
| 1202 | }; |
| 1203 | result[@backingInt(Feature.ldsbankcount32)] = .{ |
| 1204 | .llvm_name = "ldsbankcount32", |
| 1205 | .description = "The number of LDS banks per compute unit.", |
| 1206 | .dependencies = featureSet(&[_]Feature{}), |
| 1207 | }; |
| 1208 | result[@backingInt(Feature.lerp_inst)] = .{ |
| 1209 | .llvm_name = "lerp-inst", |
| 1210 | .description = "Has v_lerp_u8 instruction", |
| 1211 | .dependencies = featureSet(&[_]Feature{}), |
| 1212 | }; |
| 1213 | result[@backingInt(Feature.load_store_opt)] = .{ |
| 1214 | .llvm_name = "load-store-opt", |
| 1215 | .description = "Enable SI load/store optimizer pass", |
| 1216 | .dependencies = featureSet(&[_]Feature{}), |
| 1217 | }; |
| 1218 | result[@backingInt(Feature.lshl_add_u64_inst)] = .{ |
| 1219 | .llvm_name = "lshl-add-u64-inst", |
| 1220 | .description = "Has v_lshl_add_u64 instruction", |
| 1221 | .dependencies = featureSet(&[_]Feature{}), |
| 1222 | }; |
| 1223 | result[@backingInt(Feature.mad_intra_fwd_bug)] = .{ |
| 1224 | .llvm_name = "mad-intra-fwd-bug", |
| 1225 | .description = "MAD_U64/I64 intra instruction forwarding bug", |
| 1226 | .dependencies = featureSet(&[_]Feature{}), |
| 1227 | }; |
| 1228 | result[@backingInt(Feature.mad_mac_f32_insts)] = .{ |
| 1229 | .llvm_name = "mad-mac-f32-insts", |
| 1230 | .description = "Has v_mad_f32/v_mac_f32/v_madak_f32/v_madmk_f32 instructions", |
| 1231 | .dependencies = featureSet(&[_]Feature{}), |
| 1232 | }; |
| 1233 | result[@backingInt(Feature.mad_mix_insts)] = .{ |
| 1234 | .llvm_name = "mad-mix-insts", |
| 1235 | .description = "Has v_mad_mix_f32, v_mad_mixlo_f16, v_mad_mixhi_f16 instructions", |
| 1236 | .dependencies = featureSet(&[_]Feature{}), |
| 1237 | }; |
| 1238 | result[@backingInt(Feature.mad_u32_inst)] = .{ |
| 1239 | .llvm_name = "mad-u32-inst", |
| 1240 | .description = "Has v_mad_u32 instruction", |
| 1241 | .dependencies = featureSet(&[_]Feature{}), |
| 1242 | }; |
| 1243 | result[@backingInt(Feature.mai_insts)] = .{ |
| 1244 | .llvm_name = "mai-insts", |
| 1245 | .description = "Has mAI instructions", |
| 1246 | .dependencies = featureSet(&[_]Feature{}), |
| 1247 | }; |
| 1248 | result[@backingInt(Feature.max_hard_clause_length_32)] = .{ |
| 1249 | .llvm_name = "max-hard-clause-length-32", |
| 1250 | .description = "Maximum number of instructions in an explicit S_CLAUSE is 32", |
| 1251 | .dependencies = featureSet(&[_]Feature{}), |
| 1252 | }; |
| 1253 | result[@backingInt(Feature.max_hard_clause_length_63)] = .{ |
| 1254 | .llvm_name = "max-hard-clause-length-63", |
| 1255 | .description = "Maximum number of instructions in an explicit S_CLAUSE is 63", |
| 1256 | .dependencies = featureSet(&[_]Feature{}), |
| 1257 | }; |
| 1258 | result[@backingInt(Feature.max_private_element_size_16)] = .{ |
| 1259 | .llvm_name = "max-private-element-size-16", |
| 1260 | .description = "Maximum private access size may be 16", |
| 1261 | .dependencies = featureSet(&[_]Feature{}), |
| 1262 | }; |
| 1263 | result[@backingInt(Feature.max_private_element_size_4)] = .{ |
| 1264 | .llvm_name = "max-private-element-size-4", |
| 1265 | .description = "Maximum private access size may be 4", |
| 1266 | .dependencies = featureSet(&[_]Feature{}), |
| 1267 | }; |
| 1268 | result[@backingInt(Feature.max_private_element_size_8)] = .{ |
| 1269 | .llvm_name = "max-private-element-size-8", |
| 1270 | .description = "Maximum private access size may be 8", |
| 1271 | .dependencies = featureSet(&[_]Feature{}), |
| 1272 | }; |
| 1273 | result[@backingInt(Feature.mcast_load_insts)] = .{ |
| 1274 | .llvm_name = "mcast-load-insts", |
| 1275 | .description = "Has multicast load instructions", |
| 1276 | .dependencies = featureSet(&[_]Feature{}), |
| 1277 | }; |
| 1278 | result[@backingInt(Feature.memory_atomic_fadd_f32_denormal_support)] = .{ |
| 1279 | .llvm_name = "memory-atomic-fadd-f32-denormal-support", |
| 1280 | .description = "global/flat/buffer atomic fadd for float supports denormal handling", |
| 1281 | .dependencies = featureSet(&[_]Feature{}), |
| 1282 | }; |
| 1283 | result[@backingInt(Feature.mfma_inline_literal_bug)] = .{ |
| 1284 | .llvm_name = "mfma-inline-literal-bug", |
| 1285 | .description = "MFMA cannot use inline literal as SrcC", |
| 1286 | .dependencies = featureSet(&[_]Feature{}), |
| 1287 | }; |
| 1288 | result[@backingInt(Feature.mimg_r128)] = .{ |
| 1289 | .llvm_name = "mimg-r128", |
| 1290 | .description = "Support 128-bit texture resources", |
| 1291 | .dependencies = featureSet(&[_]Feature{}), |
| 1292 | }; |
| 1293 | result[@backingInt(Feature.min3_max3_pkf16)] = .{ |
| 1294 | .llvm_name = "min3-max3-pkf16", |
| 1295 | .description = "Has v_pk_min3_num_f16 and v_pk_max3_num_f16 instructions", |
| 1296 | .dependencies = featureSet(&[_]Feature{}), |
| 1297 | }; |
| 1298 | result[@backingInt(Feature.minimum3_maximum3_f16)] = .{ |
| 1299 | .llvm_name = "minimum3-maximum3-f16", |
| 1300 | .description = "Has v_minimum3_f16 and v_maximum3_f16 instructions", |
| 1301 | .dependencies = featureSet(&[_]Feature{}), |
| 1302 | }; |
| 1303 | result[@backingInt(Feature.minimum3_maximum3_f32)] = .{ |
| 1304 | .llvm_name = "minimum3-maximum3-f32", |
| 1305 | .description = "Has v_minimum3_f32 and v_maximum3_f32 instructions", |
| 1306 | .dependencies = featureSet(&[_]Feature{}), |
| 1307 | }; |
| 1308 | result[@backingInt(Feature.minimum3_maximum3_pkf16)] = .{ |
| 1309 | .llvm_name = "minimum3-maximum3-pkf16", |
| 1310 | .description = "Has v_pk_minimum3_f16 and v_pk_maximum3_f16 instructions", |
| 1311 | .dependencies = featureSet(&[_]Feature{}), |
| 1312 | }; |
| 1313 | result[@backingInt(Feature.movrel)] = .{ |
| 1314 | .llvm_name = "movrel", |
| 1315 | .description = "Has v_movrel*_b32 instructions", |
| 1316 | .dependencies = featureSet(&[_]Feature{}), |
| 1317 | }; |
| 1318 | result[@backingInt(Feature.msaa_load_dst_sel_bug)] = .{ |
| 1319 | .llvm_name = "msaa-load-dst-sel-bug", |
| 1320 | .description = "MSAA loads not honoring dst_sel bug", |
| 1321 | .dependencies = featureSet(&[_]Feature{}), |
| 1322 | }; |
| 1323 | result[@backingInt(Feature.negative_scratch_offset_bug)] = .{ |
| 1324 | .llvm_name = "negative-scratch-offset-bug", |
| 1325 | .description = "Negative immediate offsets in scratch instructions with an SGPR offset page fault on GFX9", |
| 1326 | .dependencies = featureSet(&[_]Feature{}), |
| 1327 | }; |
| 1328 | result[@backingInt(Feature.negative_unaligned_scratch_offset_bug)] = .{ |
| 1329 | .llvm_name = "negative-unaligned-scratch-offset-bug", |
| 1330 | .description = "Scratch instructions with a VGPR offset and a negative immediate offset that is not a multiple of 4 read wrong memory on GFX10", |
| 1331 | .dependencies = featureSet(&[_]Feature{}), |
| 1332 | }; |
| 1333 | result[@backingInt(Feature.no_data_dep_hazard)] = .{ |
| 1334 | .llvm_name = "no-data-dep-hazard", |
| 1335 | .description = "Does not need SW waitstates", |
| 1336 | .dependencies = featureSet(&[_]Feature{}), |
| 1337 | }; |
| 1338 | result[@backingInt(Feature.no_sdst_cmpx)] = .{ |
| 1339 | .llvm_name = "no-sdst-cmpx", |
| 1340 | .description = "V_CMPX does not write VCC/SGPR in addition to EXEC", |
| 1341 | .dependencies = featureSet(&[_]Feature{}), |
| 1342 | }; |
| 1343 | result[@backingInt(Feature.nsa_clause_bug)] = .{ |
| 1344 | .llvm_name = "nsa-clause-bug", |
| 1345 | .description = "MIMG-NSA in a hard clause has unpredictable results on GFX10.1", |
| 1346 | .dependencies = featureSet(&[_]Feature{}), |
| 1347 | }; |
| 1348 | result[@backingInt(Feature.nsa_encoding)] = .{ |
| 1349 | .llvm_name = "nsa-encoding", |
| 1350 | .description = "Support NSA encoding for image instructions", |
| 1351 | .dependencies = featureSet(&[_]Feature{}), |
| 1352 | }; |
| 1353 | result[@backingInt(Feature.nsa_to_vmem_bug)] = .{ |
| 1354 | .llvm_name = "nsa-to-vmem-bug", |
| 1355 | .description = "MIMG-NSA followed by VMEM fail if EXEC_LO or EXEC_HI equals zero", |
| 1356 | .dependencies = featureSet(&[_]Feature{}), |
| 1357 | }; |
| 1358 | result[@backingInt(Feature.offset_3f_bug)] = .{ |
| 1359 | .llvm_name = "offset-3f-bug", |
| 1360 | .description = "Branch offset of 3f hardware bug", |
| 1361 | .dependencies = featureSet(&[_]Feature{}), |
| 1362 | }; |
| 1363 | result[@backingInt(Feature.packed_fp32_ops)] = .{ |
| 1364 | .llvm_name = "packed-fp32-ops", |
| 1365 | .description = "Support packed fp32 instructions", |
| 1366 | .dependencies = featureSet(&[_]Feature{}), |
| 1367 | }; |
| 1368 | result[@backingInt(Feature.packed_tid)] = .{ |
| 1369 | .llvm_name = "packed-tid", |
| 1370 | .description = "Workitem IDs are packed into v0 at kernel launch", |
| 1371 | .dependencies = featureSet(&[_]Feature{}), |
| 1372 | }; |
| 1373 | result[@backingInt(Feature.partial_nsa_encoding)] = .{ |
| 1374 | .llvm_name = "partial-nsa-encoding", |
| 1375 | .description = "Support partial NSA encoding for image instructions", |
| 1376 | .dependencies = featureSet(&[_]Feature{}), |
| 1377 | }; |
| 1378 | result[@backingInt(Feature.permlane16_swap)] = .{ |
| 1379 | .llvm_name = "permlane16-swap", |
| 1380 | .description = "Has v_permlane16_swap_b32 instructions", |
| 1381 | .dependencies = featureSet(&[_]Feature{}), |
| 1382 | }; |
| 1383 | result[@backingInt(Feature.permlane32_swap)] = .{ |
| 1384 | .llvm_name = "permlane32-swap", |
| 1385 | .description = "Has v_permlane32_swap_b32 instructions", |
| 1386 | .dependencies = featureSet(&[_]Feature{}), |
| 1387 | }; |
| 1388 | result[@backingInt(Feature.pk_add_min_max_insts)] = .{ |
| 1389 | .llvm_name = "pk-add-min-max-insts", |
| 1390 | .description = "Has v_pk_add_{min|max}_{i|u}16 instructions", |
| 1391 | .dependencies = featureSet(&[_]Feature{}), |
| 1392 | }; |
| 1393 | result[@backingInt(Feature.pk_fmac_f16_inst)] = .{ |
| 1394 | .llvm_name = "pk-fmac-f16-inst", |
| 1395 | .description = "Has v_pk_fmac_f16 instruction", |
| 1396 | .dependencies = featureSet(&[_]Feature{}), |
| 1397 | }; |
| 1398 | result[@backingInt(Feature.point_sample_accel)] = .{ |
| 1399 | .llvm_name = "point-sample-accel", |
| 1400 | .description = "Has point sample acceleration feature", |
| 1401 | .dependencies = featureSet(&[_]Feature{}), |
| 1402 | }; |
| 1403 | result[@backingInt(Feature.precise_memory)] = .{ |
| 1404 | .llvm_name = "precise-memory", |
| 1405 | .description = "Enable precise memory mode", |
| 1406 | .dependencies = featureSet(&[_]Feature{}), |
| 1407 | }; |
| 1408 | result[@backingInt(Feature.priv_enabled_trap2_nop_bug)] = .{ |
| 1409 | .llvm_name = "priv-enabled-trap2-nop-bug", |
| 1410 | .description = "Hardware that runs with PRIV=1 interpreting 's_trap 2' as a nop bug", |
| 1411 | .dependencies = featureSet(&[_]Feature{}), |
| 1412 | }; |
| 1413 | result[@backingInt(Feature.prng_inst)] = .{ |
| 1414 | .llvm_name = "prng-inst", |
| 1415 | .description = "Has v_prng_b32 instruction", |
| 1416 | .dependencies = featureSet(&[_]Feature{}), |
| 1417 | }; |
| 1418 | result[@backingInt(Feature.promote_alloca)] = .{ |
| 1419 | .llvm_name = "promote-alloca", |
| 1420 | .description = "Enable promote alloca pass", |
| 1421 | .dependencies = featureSet(&[_]Feature{}), |
| 1422 | }; |
| 1423 | result[@backingInt(Feature.prt_strict_null)] = .{ |
| 1424 | .llvm_name = "enable-prt-strict-null", |
| 1425 | .description = "Enable zeroing of result registers for sparse texture fetches", |
| 1426 | .dependencies = featureSet(&[_]Feature{}), |
| 1427 | }; |
| 1428 | result[@backingInt(Feature.pseudo_scalar_trans)] = .{ |
| 1429 | .llvm_name = "pseudo-scalar-trans", |
| 1430 | .description = "Has Pseudo Scalar Transcendental instructions", |
| 1431 | .dependencies = featureSet(&[_]Feature{}), |
| 1432 | }; |
| 1433 | result[@backingInt(Feature.qsad_insts)] = .{ |
| 1434 | .llvm_name = "qsad-insts", |
| 1435 | .description = "Has v_qsad* instructions", |
| 1436 | .dependencies = featureSet(&[_]Feature{}), |
| 1437 | }; |
| 1438 | result[@backingInt(Feature.r128_a16)] = .{ |
| 1439 | .llvm_name = "r128-a16", |
| 1440 | .description = "Support gfx9-style A16 for 16-bit coordinates/gradients/lod/clamp/mip image operands, where a16 is aliased with r128", |
| 1441 | .dependencies = featureSet(&[_]Feature{}), |
| 1442 | }; |
| 1443 | result[@backingInt(Feature.real_true16)] = .{ |
| 1444 | .llvm_name = "real-true16", |
| 1445 | .description = "Use true 16-bit registers", |
| 1446 | .dependencies = featureSet(&[_]Feature{}), |
| 1447 | }; |
| 1448 | result[@backingInt(Feature.relaxed_buffer_oob_mode)] = .{ |
| 1449 | .llvm_name = "relaxed-buffer-oob-mode", |
| 1450 | .description = "Disable strict out-of-bounds buffer guarantees. An OOB access may potentially cause an adjacent access to be treated as if it were also OOB", |
| 1451 | .dependencies = featureSet(&[_]Feature{}), |
| 1452 | }; |
| 1453 | result[@backingInt(Feature.required_export_priority)] = .{ |
| 1454 | .llvm_name = "required-export-priority", |
| 1455 | .description = "Export priority must be explicitly manipulated on GFX11.5", |
| 1456 | .dependencies = featureSet(&[_]Feature{}), |
| 1457 | }; |
| 1458 | result[@backingInt(Feature.requires_cov6)] = .{ |
| 1459 | .llvm_name = "requires-cov6", |
| 1460 | .description = "Target Requires Code Object V6", |
| 1461 | .dependencies = featureSet(&[_]Feature{}), |
| 1462 | }; |
| 1463 | result[@backingInt(Feature.restricted_soffset)] = .{ |
| 1464 | .llvm_name = "restricted-soffset", |
| 1465 | .description = "Has restricted SOffset (immediate not supported).", |
| 1466 | .dependencies = featureSet(&[_]Feature{}), |
| 1467 | }; |
| 1468 | result[@backingInt(Feature.s_memrealtime)] = .{ |
| 1469 | .llvm_name = "s-memrealtime", |
| 1470 | .description = "Has s_memrealtime instruction", |
| 1471 | .dependencies = featureSet(&[_]Feature{}), |
| 1472 | }; |
| 1473 | result[@backingInt(Feature.s_memtime_inst)] = .{ |
| 1474 | .llvm_name = "s-memtime-inst", |
| 1475 | .description = "Has s_memtime instruction", |
| 1476 | .dependencies = featureSet(&[_]Feature{}), |
| 1477 | }; |
| 1478 | result[@backingInt(Feature.s_wakeup_barrier_inst)] = .{ |
| 1479 | .llvm_name = "s-wakeup-barrier-inst", |
| 1480 | .description = "Has s_wakeup_barrier instruction.", |
| 1481 | .dependencies = featureSet(&[_]Feature{}), |
| 1482 | }; |
| 1483 | result[@backingInt(Feature.sad_insts)] = .{ |
| 1484 | .llvm_name = "sad-insts", |
| 1485 | .description = "Has v_sad* instructions", |
| 1486 | .dependencies = featureSet(&[_]Feature{}), |
| 1487 | }; |
| 1488 | result[@backingInt(Feature.safe_cu_prefetch)] = .{ |
| 1489 | .llvm_name = "safe-cu-prefetch", |
| 1490 | .description = "VMEM CU scope prefetches do not fail on illegal address", |
| 1491 | .dependencies = featureSet(&[_]Feature{}), |
| 1492 | }; |
| 1493 | result[@backingInt(Feature.safe_smem_prefetch)] = .{ |
| 1494 | .llvm_name = "safe-smem-prefetch", |
| 1495 | .description = "SMEM prefetches do not fail on illegal address", |
| 1496 | .dependencies = featureSet(&[_]Feature{}), |
| 1497 | }; |
| 1498 | result[@backingInt(Feature.salu_float)] = .{ |
| 1499 | .llvm_name = "salu-float", |
| 1500 | .description = "Has SALU floating point instructions", |
| 1501 | .dependencies = featureSet(&[_]Feature{}), |
| 1502 | }; |
| 1503 | result[@backingInt(Feature.scalar_atomics)] = .{ |
| 1504 | .llvm_name = "scalar-atomics", |
| 1505 | .description = "Has atomic scalar memory instructions", |
| 1506 | .dependencies = featureSet(&[_]Feature{}), |
| 1507 | }; |
| 1508 | result[@backingInt(Feature.scalar_dwordx3_loads)] = .{ |
| 1509 | .llvm_name = "scalar-dwordx3-loads", |
| 1510 | .description = "Has 96-bit scalar load instructions", |
| 1511 | .dependencies = featureSet(&[_]Feature{}), |
| 1512 | }; |
| 1513 | result[@backingInt(Feature.scalar_flat_scratch_insts)] = .{ |
| 1514 | .llvm_name = "scalar-flat-scratch-insts", |
| 1515 | .description = "Have s_scratch_* flat memory instructions", |
| 1516 | .dependencies = featureSet(&[_]Feature{}), |
| 1517 | }; |
| 1518 | result[@backingInt(Feature.scalar_stores)] = .{ |
| 1519 | .llvm_name = "scalar-stores", |
| 1520 | .description = "Has store scalar memory instructions", |
| 1521 | .dependencies = featureSet(&[_]Feature{}), |
| 1522 | }; |
| 1523 | result[@backingInt(Feature.sdwa)] = .{ |
| 1524 | .llvm_name = "sdwa", |
| 1525 | .description = "Support SDWA (Sub-DWORD Addressing) extension", |
| 1526 | .dependencies = featureSet(&[_]Feature{}), |
| 1527 | }; |
| 1528 | result[@backingInt(Feature.sdwa_mav)] = .{ |
| 1529 | .llvm_name = "sdwa-mav", |
| 1530 | .description = "Support v_mac_f32/f16 with SDWA (Sub-DWORD Addressing) extension", |
| 1531 | .dependencies = featureSet(&[_]Feature{}), |
| 1532 | }; |
| 1533 | result[@backingInt(Feature.sdwa_omod)] = .{ |
| 1534 | .llvm_name = "sdwa-omod", |
| 1535 | .description = "Support OMod with SDWA (Sub-DWORD Addressing) extension", |
| 1536 | .dependencies = featureSet(&[_]Feature{}), |
| 1537 | }; |
| 1538 | result[@backingInt(Feature.sdwa_out_mods_vopc)] = .{ |
| 1539 | .llvm_name = "sdwa-out-mods-vopc", |
| 1540 | .description = "Support clamp for VOPC with SDWA (Sub-DWORD Addressing) extension", |
| 1541 | .dependencies = featureSet(&[_]Feature{}), |
| 1542 | }; |
| 1543 | result[@backingInt(Feature.sdwa_scalar)] = .{ |
| 1544 | .llvm_name = "sdwa-scalar", |
| 1545 | .description = "Support scalar register with SDWA (Sub-DWORD Addressing) extension", |
| 1546 | .dependencies = featureSet(&[_]Feature{}), |
| 1547 | }; |
| 1548 | result[@backingInt(Feature.sdwa_sdst)] = .{ |
| 1549 | .llvm_name = "sdwa-sdst", |
| 1550 | .description = "Support scalar dst for VOPC with SDWA (Sub-DWORD Addressing) extension", |
| 1551 | .dependencies = featureSet(&[_]Feature{}), |
| 1552 | }; |
| 1553 | result[@backingInt(Feature.sea_islands)] = .{ |
| 1554 | .llvm_name = "sea-islands", |
| 1555 | .description = "SEA_ISLANDS GPU generation", |
| 1556 | .dependencies = featureSet(&[_]Feature{ |
| 1557 | .addressablelocalmemorysize65536, |
| 1558 | .atomic_fmin_fmax_flat_f32, |
| 1559 | .atomic_fmin_fmax_flat_f64, |
| 1560 | .atomic_fmin_fmax_global_f32, |
| 1561 | .atomic_fmin_fmax_global_f64, |
| 1562 | .ci_insts, |
| 1563 | .cube_insts, |
| 1564 | .cvt_pknorm_vop2_insts, |
| 1565 | .default_component_zero, |
| 1566 | .ds_src2_insts, |
| 1567 | .extended_image_insts, |
| 1568 | .fp64, |
| 1569 | .gds, |
| 1570 | .gfx7_gfx8_gfx9_insts, |
| 1571 | .gws, |
| 1572 | .image_insts, |
| 1573 | .lerp_inst, |
| 1574 | .mad_mac_f32_insts, |
| 1575 | .mimg_r128, |
| 1576 | .movrel, |
| 1577 | .qsad_insts, |
| 1578 | .s_memtime_inst, |
| 1579 | .sad_insts, |
| 1580 | .trig_reduced_range, |
| 1581 | .unaligned_buffer_access, |
| 1582 | .vmem_write_vgpr_in_order, |
| 1583 | .wavefrontsize64, |
| 1584 | }), |
| 1585 | }; |
| 1586 | result[@backingInt(Feature.setprio_inc_wg_inst)] = .{ |
| 1587 | .llvm_name = "setprio-inc-wg-inst", |
| 1588 | .description = "Has s_setprio_inc_wg instruction.", |
| 1589 | .dependencies = featureSet(&[_]Feature{}), |
| 1590 | }; |
| 1591 | result[@backingInt(Feature.setreg_vgpr_msb_fixup)] = .{ |
| 1592 | .llvm_name = "setreg-vgpr-msb-fixup", |
| 1593 | .description = "S_SETREG to MODE clobbers VGPR MSB bits, requires fixup", |
| 1594 | .dependencies = featureSet(&[_]Feature{}), |
| 1595 | }; |
| 1596 | result[@backingInt(Feature.sgpr_init_bug)] = .{ |
| 1597 | .llvm_name = "sgpr-init-bug", |
| 1598 | .description = "VI SGPR initialization bug requiring a fixed SGPR allocation size", |
| 1599 | .dependencies = featureSet(&[_]Feature{}), |
| 1600 | }; |
| 1601 | result[@backingInt(Feature.shader_cycles_hi_lo_registers)] = .{ |
| 1602 | .llvm_name = "shader-cycles-hi-lo-registers", |
| 1603 | .description = "Has SHADER_CYCLES_HI/LO hardware registers", |
| 1604 | .dependencies = featureSet(&[_]Feature{}), |
| 1605 | }; |
| 1606 | result[@backingInt(Feature.shader_cycles_register)] = .{ |
| 1607 | .llvm_name = "shader-cycles-register", |
| 1608 | .description = "Has SHADER_CYCLES hardware register", |
| 1609 | .dependencies = featureSet(&[_]Feature{}), |
| 1610 | }; |
| 1611 | result[@backingInt(Feature.si_scheduler)] = .{ |
| 1612 | .llvm_name = "si-scheduler", |
| 1613 | .description = "Enable SI Machine Scheduler", |
| 1614 | .dependencies = featureSet(&[_]Feature{}), |
| 1615 | }; |
| 1616 | result[@backingInt(Feature.smem_to_vector_write_hazard)] = .{ |
| 1617 | .llvm_name = "smem-to-vector-write-hazard", |
| 1618 | .description = "s_load_dword followed by v_cmp page faults", |
| 1619 | .dependencies = featureSet(&[_]Feature{}), |
| 1620 | }; |
| 1621 | result[@backingInt(Feature.southern_islands)] = .{ |
| 1622 | .llvm_name = "southern-islands", |
| 1623 | .description = "SOUTHERN_ISLANDS GPU generation", |
| 1624 | .dependencies = featureSet(&[_]Feature{ |
| 1625 | .addressablelocalmemorysize32768, |
| 1626 | .atomic_fmin_fmax_global_f32, |
| 1627 | .atomic_fmin_fmax_global_f64, |
| 1628 | .cube_insts, |
| 1629 | .cvt_pknorm_vop2_insts, |
| 1630 | .default_component_zero, |
| 1631 | .ds_src2_insts, |
| 1632 | .extended_image_insts, |
| 1633 | .fp64, |
| 1634 | .gds, |
| 1635 | .gws, |
| 1636 | .image_insts, |
| 1637 | .ldsbankcount32, |
| 1638 | .lerp_inst, |
| 1639 | .mad_mac_f32_insts, |
| 1640 | .mimg_r128, |
| 1641 | .movrel, |
| 1642 | .s_memtime_inst, |
| 1643 | .sad_insts, |
| 1644 | .trig_reduced_range, |
| 1645 | .vmem_write_vgpr_in_order, |
| 1646 | .wavefrontsize64, |
| 1647 | }), |
| 1648 | }; |
| 1649 | result[@backingInt(Feature.sramecc)] = .{ |
| 1650 | .llvm_name = "sramecc", |
| 1651 | .description = "Enable SRAMECC", |
| 1652 | .dependencies = featureSet(&[_]Feature{}), |
| 1653 | }; |
| 1654 | result[@backingInt(Feature.sramecc_support)] = .{ |
| 1655 | .llvm_name = "sramecc-support", |
| 1656 | .description = "Hardware supports SRAMECC", |
| 1657 | .dependencies = featureSet(&[_]Feature{}), |
| 1658 | }; |
| 1659 | result[@backingInt(Feature.tanh_insts)] = .{ |
| 1660 | .llvm_name = "tanh-insts", |
| 1661 | .description = "Has v_tanh_f32/f16 instructions", |
| 1662 | .dependencies = featureSet(&[_]Feature{}), |
| 1663 | }; |
| 1664 | result[@backingInt(Feature.tensor_cvt_lut_insts)] = .{ |
| 1665 | .llvm_name = "tensor-cvt-lut-insts", |
| 1666 | .description = "Has v_perm_pk16* instructions", |
| 1667 | .dependencies = featureSet(&[_]Feature{}), |
| 1668 | }; |
| 1669 | result[@backingInt(Feature.tgsplit)] = .{ |
| 1670 | .llvm_name = "tgsplit", |
| 1671 | .description = "Enable threadgroup split execution", |
| 1672 | .dependencies = featureSet(&[_]Feature{}), |
| 1673 | }; |
| 1674 | result[@backingInt(Feature.transpose_load_f4f6_insts)] = .{ |
| 1675 | .llvm_name = "transpose-load-f4f6-insts", |
| 1676 | .description = "Has ds_load_tr4/tr6 and global_load_tr4/tr6 instructions", |
| 1677 | .dependencies = featureSet(&[_]Feature{}), |
| 1678 | }; |
| 1679 | result[@backingInt(Feature.trap_handler)] = .{ |
| 1680 | .llvm_name = "trap-handler", |
| 1681 | .description = "Trap handler support", |
| 1682 | .dependencies = featureSet(&[_]Feature{}), |
| 1683 | }; |
| 1684 | result[@backingInt(Feature.trig_reduced_range)] = .{ |
| 1685 | .llvm_name = "trig-reduced-range", |
| 1686 | .description = "Requires use of fract on arguments to trig instructions", |
| 1687 | .dependencies = featureSet(&[_]Feature{}), |
| 1688 | }; |
| 1689 | result[@backingInt(Feature.true16)] = .{ |
| 1690 | .llvm_name = "true16", |
| 1691 | .description = "True 16-bit operand instructions", |
| 1692 | .dependencies = featureSet(&[_]Feature{}), |
| 1693 | }; |
| 1694 | result[@backingInt(Feature.unaligned_access_mode)] = .{ |
| 1695 | .llvm_name = "unaligned-access-mode", |
| 1696 | .description = "Enable unaligned global, local and region loads and stores if the hardware supports it", |
| 1697 | .dependencies = featureSet(&[_]Feature{}), |
| 1698 | }; |
| 1699 | result[@backingInt(Feature.unaligned_buffer_access)] = .{ |
| 1700 | .llvm_name = "unaligned-buffer-access", |
| 1701 | .description = "Hardware supports unaligned global loads and stores", |
| 1702 | .dependencies = featureSet(&[_]Feature{}), |
| 1703 | }; |
| 1704 | result[@backingInt(Feature.unaligned_ds_access)] = .{ |
| 1705 | .llvm_name = "unaligned-ds-access", |
| 1706 | .description = "Hardware supports unaligned local and region loads and stores", |
| 1707 | .dependencies = featureSet(&[_]Feature{}), |
| 1708 | }; |
| 1709 | result[@backingInt(Feature.unaligned_scratch_access)] = .{ |
| 1710 | .llvm_name = "unaligned-scratch-access", |
| 1711 | .description = "Support unaligned scratch loads and stores", |
| 1712 | .dependencies = featureSet(&[_]Feature{}), |
| 1713 | }; |
| 1714 | result[@backingInt(Feature.unpacked_d16_vmem)] = .{ |
| 1715 | .llvm_name = "unpacked-d16-vmem", |
| 1716 | .description = "Has unpacked d16 vmem instructions", |
| 1717 | .dependencies = featureSet(&[_]Feature{}), |
| 1718 | }; |
| 1719 | result[@backingInt(Feature.unsafe_ds_offset_folding)] = .{ |
| 1720 | .llvm_name = "unsafe-ds-offset-folding", |
| 1721 | .description = "Force using DS instruction immediate offsets on SI", |
| 1722 | .dependencies = featureSet(&[_]Feature{}), |
| 1723 | }; |
| 1724 | result[@backingInt(Feature.user_sgpr_init16_bug)] = .{ |
| 1725 | .llvm_name = "user-sgpr-init16-bug", |
| 1726 | .description = "Bug requiring at least 16 user+system SGPRs to be enabled", |
| 1727 | .dependencies = featureSet(&[_]Feature{}), |
| 1728 | }; |
| 1729 | result[@backingInt(Feature.valu_trans_use_hazard)] = .{ |
| 1730 | .llvm_name = "valu-trans-use-hazard", |
| 1731 | .description = "Hazard when TRANS instructions are closely followed by a use of the result", |
| 1732 | .dependencies = featureSet(&[_]Feature{}), |
| 1733 | }; |
| 1734 | result[@backingInt(Feature.vcmpx_exec_war_hazard)] = .{ |
| 1735 | .llvm_name = "vcmpx-exec-war-hazard", |
| 1736 | .description = "V_CMPX WAR hazard on EXEC (V_CMPX issue ONLY)", |
| 1737 | .dependencies = featureSet(&[_]Feature{}), |
| 1738 | }; |
| 1739 | result[@backingInt(Feature.vcmpx_permlane_hazard)] = .{ |
| 1740 | .llvm_name = "vcmpx-permlane-hazard", |
| 1741 | .description = "TODO: describe me", |
| 1742 | .dependencies = featureSet(&[_]Feature{}), |
| 1743 | }; |
| 1744 | result[@backingInt(Feature.vgpr_align2)] = .{ |
| 1745 | .llvm_name = "vgpr-align2", |
| 1746 | .description = "VGPR and AGPR tuple operands require even alignment", |
| 1747 | .dependencies = featureSet(&[_]Feature{}), |
| 1748 | }; |
| 1749 | result[@backingInt(Feature.vgpr_index_mode)] = .{ |
| 1750 | .llvm_name = "vgpr-index-mode", |
| 1751 | .description = "Has VGPR mode register indexing", |
| 1752 | .dependencies = featureSet(&[_]Feature{}), |
| 1753 | }; |
| 1754 | result[@backingInt(Feature.vmem_pref_insts)] = .{ |
| 1755 | .llvm_name = "vmem-pref-insts", |
| 1756 | .description = "Has flat_prefect_b8 and global_prefetch_b8 instructions", |
| 1757 | .dependencies = featureSet(&[_]Feature{}), |
| 1758 | }; |
| 1759 | result[@backingInt(Feature.vmem_to_lds_load_insts)] = .{ |
| 1760 | .llvm_name = "vmem-to-lds-load-insts", |
| 1761 | .description = "The platform has memory to lds instructions (global_load w/lds bit set, buffer_load w/lds bit set or global_load_lds. This does not include scratch_load_lds.", |
| 1762 | .dependencies = featureSet(&[_]Feature{}), |
| 1763 | }; |
| 1764 | result[@backingInt(Feature.vmem_to_scalar_write_hazard)] = .{ |
| 1765 | .llvm_name = "vmem-to-scalar-write-hazard", |
| 1766 | .description = "VMEM instruction followed by scalar writing to EXEC mask, M0 or SGPR leads to incorrect execution.", |
| 1767 | .dependencies = featureSet(&[_]Feature{}), |
| 1768 | }; |
| 1769 | result[@backingInt(Feature.vmem_write_vgpr_in_order)] = .{ |
| 1770 | .llvm_name = "vmem-write-vgpr-in-order", |
| 1771 | .description = "VMEM instructions of the same type write VGPR results in order", |
| 1772 | .dependencies = featureSet(&[_]Feature{}), |
| 1773 | }; |
| 1774 | result[@backingInt(Feature.volcanic_islands)] = .{ |
| 1775 | .llvm_name = "volcanic-islands", |
| 1776 | .description = "VOLCANIC_ISLANDS GPU generation", |
| 1777 | .dependencies = featureSet(&[_]Feature{ |
| 1778 | .@"16_bit_insts", |
| 1779 | .addressablelocalmemorysize65536, |
| 1780 | .ci_insts, |
| 1781 | .cube_insts, |
| 1782 | .cvt_pknorm_vop2_insts, |
| 1783 | .default_component_zero, |
| 1784 | .dpp, |
| 1785 | .ds_src2_insts, |
| 1786 | .extended_image_insts, |
| 1787 | .fast_denormal_f32, |
| 1788 | .flat_address_space, |
| 1789 | .fp64, |
| 1790 | .gcn3_encoding, |
| 1791 | .gds, |
| 1792 | .gfx7_gfx8_gfx9_insts, |
| 1793 | .gfx8_insts, |
| 1794 | .gws, |
| 1795 | .image_insts, |
| 1796 | .int_clamp_insts, |
| 1797 | .inv_2pi_inline_imm, |
| 1798 | .lerp_inst, |
| 1799 | .mad_mac_f32_insts, |
| 1800 | .mimg_r128, |
| 1801 | .movrel, |
| 1802 | .qsad_insts, |
| 1803 | .s_memrealtime, |
| 1804 | .s_memtime_inst, |
| 1805 | .sad_insts, |
| 1806 | .scalar_stores, |
| 1807 | .sdwa, |
| 1808 | .sdwa_mav, |
| 1809 | .sdwa_out_mods_vopc, |
| 1810 | .trig_reduced_range, |
| 1811 | .unaligned_buffer_access, |
| 1812 | .vgpr_index_mode, |
| 1813 | .vmem_write_vgpr_in_order, |
| 1814 | .wavefrontsize64, |
| 1815 | }), |
| 1816 | }; |
| 1817 | result[@backingInt(Feature.vop3_literal)] = .{ |
| 1818 | .llvm_name = "vop3-literal", |
| 1819 | .description = "Can use one literal in VOP3", |
| 1820 | .dependencies = featureSet(&[_]Feature{}), |
| 1821 | }; |
| 1822 | result[@backingInt(Feature.vop3p)] = .{ |
| 1823 | .llvm_name = "vop3p", |
| 1824 | .description = "Has VOP3P packed instructions", |
| 1825 | .dependencies = featureSet(&[_]Feature{}), |
| 1826 | }; |
| 1827 | result[@backingInt(Feature.vopd)] = .{ |
| 1828 | .llvm_name = "vopd", |
| 1829 | .description = "Has VOPD dual issue wave32 instructions", |
| 1830 | .dependencies = featureSet(&[_]Feature{}), |
| 1831 | }; |
| 1832 | result[@backingInt(Feature.vscnt)] = .{ |
| 1833 | .llvm_name = "vscnt", |
| 1834 | .description = "Has separate store vscnt counter", |
| 1835 | .dependencies = featureSet(&[_]Feature{}), |
| 1836 | }; |
| 1837 | result[@backingInt(Feature.wait_xcnt)] = .{ |
| 1838 | .llvm_name = "wait-xcnt", |
| 1839 | .description = "Has s_wait_xcnt instruction", |
| 1840 | .dependencies = featureSet(&[_]Feature{}), |
| 1841 | }; |
| 1842 | result[@backingInt(Feature.waits_before_system_scope_stores)] = .{ |
| 1843 | .llvm_name = "waits-before-system-scope-stores", |
| 1844 | .description = "Target requires waits for loads and atomics before system scope stores", |
| 1845 | .dependencies = featureSet(&[_]Feature{}), |
| 1846 | }; |
| 1847 | result[@backingInt(Feature.wavefrontsize16)] = .{ |
| 1848 | .llvm_name = "wavefrontsize16", |
| 1849 | .description = "The number of threads per wavefront", |
| 1850 | .dependencies = featureSet(&[_]Feature{}), |
| 1851 | }; |
| 1852 | result[@backingInt(Feature.wavefrontsize32)] = .{ |
| 1853 | .llvm_name = "wavefrontsize32", |
| 1854 | .description = "The number of threads per wavefront", |
| 1855 | .dependencies = featureSet(&[_]Feature{}), |
| 1856 | }; |
| 1857 | result[@backingInt(Feature.wavefrontsize64)] = .{ |
| 1858 | .llvm_name = "wavefrontsize64", |
| 1859 | .description = "The number of threads per wavefront", |
| 1860 | .dependencies = featureSet(&[_]Feature{}), |
| 1861 | }; |
| 1862 | result[@backingInt(Feature.xf32_insts)] = .{ |
| 1863 | .llvm_name = "xf32-insts", |
| 1864 | .description = "Has instructions that support xf32 format, such as v_mfma_f32_16x16x8_xf32 and v_mfma_f32_32x32x4_xf32", |
| 1865 | .dependencies = featureSet(&[_]Feature{}), |
| 1866 | }; |
| 1867 | result[@backingInt(Feature.xnack)] = .{ |
| 1868 | .llvm_name = "xnack", |
| 1869 | .description = "Enable XNACK support", |
| 1870 | .dependencies = featureSet(&[_]Feature{}), |
| 1871 | }; |
| 1872 | result[@backingInt(Feature.xnack_support)] = .{ |
| 1873 | .llvm_name = "xnack-support", |
| 1874 | .description = "Hardware supports XNACK", |
| 1875 | .dependencies = featureSet(&[_]Feature{}), |
| 1876 | }; |
| 1877 | const ti = @typeInfo(Feature); |
| 1878 | for (&result, 0..) |*elem, i| { |
| 1879 | elem.index = i; |
| 1880 | elem.name = ti.@"enum".field_names[i]; |
| 1881 | } |
| 1882 | break :blk result; |
| 1883 | }; |
| 1884 | |
| 1885 | pub const cpu = struct { |
| 1886 | pub const bonaire: CpuModel = .{ |
| 1887 | .name = "bonaire", |
| 1888 | .llvm_name = "bonaire", |
| 1889 | .features = featureSet(&[_]Feature{ |
| 1890 | .ldsbankcount32, |
| 1891 | .sea_islands, |
| 1892 | }), |
| 1893 | }; |
| 1894 | pub const carrizo: CpuModel = .{ |
| 1895 | .name = "carrizo", |
| 1896 | .llvm_name = "carrizo", |
| 1897 | .features = featureSet(&[_]Feature{ |
| 1898 | .fast_fmaf, |
| 1899 | .half_rate_64_ops, |
| 1900 | .ldsbankcount32, |
| 1901 | .unpacked_d16_vmem, |
| 1902 | .volcanic_islands, |
| 1903 | .xnack_support, |
| 1904 | }), |
| 1905 | }; |
| 1906 | pub const fiji: CpuModel = .{ |
| 1907 | .name = "fiji", |
| 1908 | .llvm_name = "fiji", |
| 1909 | .features = featureSet(&[_]Feature{ |
| 1910 | .ldsbankcount32, |
| 1911 | .unpacked_d16_vmem, |
| 1912 | .volcanic_islands, |
| 1913 | }), |
| 1914 | }; |
| 1915 | pub const generic: CpuModel = .{ |
| 1916 | .name = "generic", |
| 1917 | .llvm_name = "generic", |
| 1918 | .features = featureSet(&[_]Feature{}), |
| 1919 | }; |
| 1920 | pub const generic_hsa: CpuModel = .{ |
| 1921 | .name = "generic_hsa", |
| 1922 | .llvm_name = "generic-hsa", |
| 1923 | .features = featureSet(&[_]Feature{ |
| 1924 | .flat_address_space, |
| 1925 | }), |
| 1926 | }; |
| 1927 | pub const gfx1010: CpuModel = .{ |
| 1928 | .name = "gfx1010", |
| 1929 | .llvm_name = "gfx1010", |
| 1930 | .features = featureSet(&[_]Feature{ |
| 1931 | .back_off_barrier, |
| 1932 | .dl_insts, |
| 1933 | .ds_src2_insts, |
| 1934 | .flat_segment_offset_bug, |
| 1935 | .get_wave_id_inst, |
| 1936 | .gfx10, |
| 1937 | .inst_fwd_prefetch_bug, |
| 1938 | .lds_branch_vmem_war_hazard, |
| 1939 | .lds_misaligned_bug, |
| 1940 | .ldsbankcount32, |
| 1941 | .mad_mac_f32_insts, |
| 1942 | .negative_unaligned_scratch_offset_bug, |
| 1943 | .nsa_clause_bug, |
| 1944 | .nsa_encoding, |
| 1945 | .nsa_to_vmem_bug, |
| 1946 | .offset_3f_bug, |
| 1947 | .scalar_atomics, |
| 1948 | .scalar_flat_scratch_insts, |
| 1949 | .scalar_stores, |
| 1950 | .smem_to_vector_write_hazard, |
| 1951 | .vcmpx_exec_war_hazard, |
| 1952 | .vcmpx_permlane_hazard, |
| 1953 | .vmem_to_scalar_write_hazard, |
| 1954 | .xnack_support, |
| 1955 | }), |
| 1956 | }; |
| 1957 | pub const gfx1011: CpuModel = .{ |
| 1958 | .name = "gfx1011", |
| 1959 | .llvm_name = "gfx1011", |
| 1960 | .features = featureSet(&[_]Feature{ |
| 1961 | .back_off_barrier, |
| 1962 | .dl_insts, |
| 1963 | .dot10_insts, |
| 1964 | .dot1_insts, |
| 1965 | .dot2_insts, |
| 1966 | .dot5_insts, |
| 1967 | .dot6_insts, |
| 1968 | .dot7_insts, |
| 1969 | .ds_src2_insts, |
| 1970 | .flat_segment_offset_bug, |
| 1971 | .get_wave_id_inst, |
| 1972 | .gfx10, |
| 1973 | .inst_fwd_prefetch_bug, |
| 1974 | .lds_branch_vmem_war_hazard, |
| 1975 | .lds_misaligned_bug, |
| 1976 | .ldsbankcount32, |
| 1977 | .mad_mac_f32_insts, |
| 1978 | .negative_unaligned_scratch_offset_bug, |
| 1979 | .nsa_clause_bug, |
| 1980 | .nsa_encoding, |
| 1981 | .nsa_to_vmem_bug, |
| 1982 | .offset_3f_bug, |
| 1983 | .scalar_atomics, |
| 1984 | .scalar_flat_scratch_insts, |
| 1985 | .scalar_stores, |
| 1986 | .smem_to_vector_write_hazard, |
| 1987 | .vcmpx_exec_war_hazard, |
| 1988 | .vcmpx_permlane_hazard, |
| 1989 | .vmem_to_scalar_write_hazard, |
| 1990 | .xnack_support, |
| 1991 | }), |
| 1992 | }; |
| 1993 | pub const gfx1012: CpuModel = .{ |
| 1994 | .name = "gfx1012", |
| 1995 | .llvm_name = "gfx1012", |
| 1996 | .features = featureSet(&[_]Feature{ |
| 1997 | .back_off_barrier, |
| 1998 | .dl_insts, |
| 1999 | .dot10_insts, |
| 2000 | .dot1_insts, |
| 2001 | .dot2_insts, |
| 2002 | .dot5_insts, |
| 2003 | .dot6_insts, |
| 2004 | .dot7_insts, |
| 2005 | .ds_src2_insts, |
| 2006 | .flat_segment_offset_bug, |
| 2007 | .get_wave_id_inst, |
| 2008 | .gfx10, |
| 2009 | .inst_fwd_prefetch_bug, |
| 2010 | .lds_branch_vmem_war_hazard, |
| 2011 | .lds_misaligned_bug, |
| 2012 | .ldsbankcount32, |
| 2013 | .mad_mac_f32_insts, |
| 2014 | .negative_unaligned_scratch_offset_bug, |
| 2015 | .nsa_clause_bug, |
| 2016 | .nsa_encoding, |
| 2017 | .nsa_to_vmem_bug, |
| 2018 | .offset_3f_bug, |
| 2019 | .scalar_atomics, |
| 2020 | .scalar_flat_scratch_insts, |
| 2021 | .scalar_stores, |
| 2022 | .smem_to_vector_write_hazard, |
| 2023 | .vcmpx_exec_war_hazard, |
| 2024 | .vcmpx_permlane_hazard, |
| 2025 | .vmem_to_scalar_write_hazard, |
| 2026 | .xnack_support, |
| 2027 | }), |
| 2028 | }; |
| 2029 | pub const gfx1013: CpuModel = .{ |
| 2030 | .name = "gfx1013", |
| 2031 | .llvm_name = "gfx1013", |
| 2032 | .features = featureSet(&[_]Feature{ |
| 2033 | .back_off_barrier, |
| 2034 | .dl_insts, |
| 2035 | .ds_src2_insts, |
| 2036 | .flat_segment_offset_bug, |
| 2037 | .get_wave_id_inst, |
| 2038 | .gfx10, |
| 2039 | .gfx10_a_encoding, |
| 2040 | .inst_fwd_prefetch_bug, |
| 2041 | .lds_branch_vmem_war_hazard, |
| 2042 | .lds_misaligned_bug, |
| 2043 | .ldsbankcount32, |
| 2044 | .mad_mac_f32_insts, |
| 2045 | .negative_unaligned_scratch_offset_bug, |
| 2046 | .nsa_clause_bug, |
| 2047 | .nsa_encoding, |
| 2048 | .nsa_to_vmem_bug, |
| 2049 | .offset_3f_bug, |
| 2050 | .scalar_atomics, |
| 2051 | .scalar_flat_scratch_insts, |
| 2052 | .scalar_stores, |
| 2053 | .smem_to_vector_write_hazard, |
| 2054 | .vcmpx_exec_war_hazard, |
| 2055 | .vcmpx_permlane_hazard, |
| 2056 | .vmem_to_scalar_write_hazard, |
| 2057 | .xnack_support, |
| 2058 | }), |
| 2059 | }; |
| 2060 | pub const gfx1030: CpuModel = .{ |
| 2061 | .name = "gfx1030", |
| 2062 | .llvm_name = "gfx1030", |
| 2063 | .features = featureSet(&[_]Feature{ |
| 2064 | .back_off_barrier, |
| 2065 | .dl_insts, |
| 2066 | .dot10_insts, |
| 2067 | .dot1_insts, |
| 2068 | .dot2_insts, |
| 2069 | .dot5_insts, |
| 2070 | .dot6_insts, |
| 2071 | .dot7_insts, |
| 2072 | .gfx10, |
| 2073 | .gfx10_3_insts, |
| 2074 | .gfx10_a_encoding, |
| 2075 | .gfx10_b_encoding, |
| 2076 | .ldsbankcount32, |
| 2077 | .nsa_encoding, |
| 2078 | .shader_cycles_register, |
| 2079 | }), |
| 2080 | }; |
| 2081 | pub const gfx1031: CpuModel = .{ |
| 2082 | .name = "gfx1031", |
| 2083 | .llvm_name = "gfx1031", |
| 2084 | .features = featureSet(&[_]Feature{ |
| 2085 | .back_off_barrier, |
| 2086 | .dl_insts, |
| 2087 | .dot10_insts, |
| 2088 | .dot1_insts, |
| 2089 | .dot2_insts, |
| 2090 | .dot5_insts, |
| 2091 | .dot6_insts, |
| 2092 | .dot7_insts, |
| 2093 | .gfx10, |
| 2094 | .gfx10_3_insts, |
| 2095 | .gfx10_a_encoding, |
| 2096 | .gfx10_b_encoding, |
| 2097 | .ldsbankcount32, |
| 2098 | .nsa_encoding, |
| 2099 | .shader_cycles_register, |
| 2100 | }), |
| 2101 | }; |
| 2102 | pub const gfx1032: CpuModel = .{ |
| 2103 | .name = "gfx1032", |
| 2104 | .llvm_name = "gfx1032", |
| 2105 | .features = featureSet(&[_]Feature{ |
| 2106 | .back_off_barrier, |
| 2107 | .dl_insts, |
| 2108 | .dot10_insts, |
| 2109 | .dot1_insts, |
| 2110 | .dot2_insts, |
| 2111 | .dot5_insts, |
| 2112 | .dot6_insts, |
| 2113 | .dot7_insts, |
| 2114 | .gfx10, |
| 2115 | .gfx10_3_insts, |
| 2116 | .gfx10_a_encoding, |
| 2117 | .gfx10_b_encoding, |
| 2118 | .ldsbankcount32, |
| 2119 | .nsa_encoding, |
| 2120 | .shader_cycles_register, |
| 2121 | }), |
| 2122 | }; |
| 2123 | pub const gfx1033: CpuModel = .{ |
| 2124 | .name = "gfx1033", |
| 2125 | .llvm_name = "gfx1033", |
| 2126 | .features = featureSet(&[_]Feature{ |
| 2127 | .back_off_barrier, |
| 2128 | .dl_insts, |
| 2129 | .dot10_insts, |
| 2130 | .dot1_insts, |
| 2131 | .dot2_insts, |
| 2132 | .dot5_insts, |
| 2133 | .dot6_insts, |
| 2134 | .dot7_insts, |
| 2135 | .gfx10, |
| 2136 | .gfx10_3_insts, |
| 2137 | .gfx10_a_encoding, |
| 2138 | .gfx10_b_encoding, |
| 2139 | .ldsbankcount32, |
| 2140 | .nsa_encoding, |
| 2141 | .shader_cycles_register, |
| 2142 | }), |
| 2143 | }; |
| 2144 | pub const gfx1034: CpuModel = .{ |
| 2145 | .name = "gfx1034", |
| 2146 | .llvm_name = "gfx1034", |
| 2147 | .features = featureSet(&[_]Feature{ |
| 2148 | .back_off_barrier, |
| 2149 | .dl_insts, |
| 2150 | .dot10_insts, |
| 2151 | .dot1_insts, |
| 2152 | .dot2_insts, |
| 2153 | .dot5_insts, |
| 2154 | .dot6_insts, |
| 2155 | .dot7_insts, |
| 2156 | .gfx10, |
| 2157 | .gfx10_3_insts, |
| 2158 | .gfx10_a_encoding, |
| 2159 | .gfx10_b_encoding, |
| 2160 | .ldsbankcount32, |
| 2161 | .nsa_encoding, |
| 2162 | .shader_cycles_register, |
| 2163 | }), |
| 2164 | }; |
| 2165 | pub const gfx1035: CpuModel = .{ |
| 2166 | .name = "gfx1035", |
| 2167 | .llvm_name = "gfx1035", |
| 2168 | .features = featureSet(&[_]Feature{ |
| 2169 | .back_off_barrier, |
| 2170 | .dl_insts, |
| 2171 | .dot10_insts, |
| 2172 | .dot1_insts, |
| 2173 | .dot2_insts, |
| 2174 | .dot5_insts, |
| 2175 | .dot6_insts, |
| 2176 | .dot7_insts, |
| 2177 | .gfx10, |
| 2178 | .gfx10_3_insts, |
| 2179 | .gfx10_a_encoding, |
| 2180 | .gfx10_b_encoding, |
| 2181 | .ldsbankcount32, |
| 2182 | .nsa_encoding, |
| 2183 | .shader_cycles_register, |
| 2184 | }), |
| 2185 | }; |
| 2186 | pub const gfx1036: CpuModel = .{ |
| 2187 | .name = "gfx1036", |
| 2188 | .llvm_name = "gfx1036", |
| 2189 | .features = featureSet(&[_]Feature{ |
| 2190 | .back_off_barrier, |
| 2191 | .dl_insts, |
| 2192 | .dot10_insts, |
| 2193 | .dot1_insts, |
| 2194 | .dot2_insts, |
| 2195 | .dot5_insts, |
| 2196 | .dot6_insts, |
| 2197 | .dot7_insts, |
| 2198 | .gfx10, |
| 2199 | .gfx10_3_insts, |
| 2200 | .gfx10_a_encoding, |
| 2201 | .gfx10_b_encoding, |
| 2202 | .ldsbankcount32, |
| 2203 | .nsa_encoding, |
| 2204 | .shader_cycles_register, |
| 2205 | }), |
| 2206 | }; |
| 2207 | pub const gfx10_1_generic: CpuModel = .{ |
| 2208 | .name = "gfx10_1_generic", |
| 2209 | .llvm_name = "gfx10-1-generic", |
| 2210 | .features = featureSet(&[_]Feature{ |
| 2211 | .back_off_barrier, |
| 2212 | .dl_insts, |
| 2213 | .ds_src2_insts, |
| 2214 | .flat_segment_offset_bug, |
| 2215 | .get_wave_id_inst, |
| 2216 | .gfx10, |
| 2217 | .inst_fwd_prefetch_bug, |
| 2218 | .lds_branch_vmem_war_hazard, |
| 2219 | .lds_misaligned_bug, |
| 2220 | .ldsbankcount32, |
| 2221 | .mad_mac_f32_insts, |
| 2222 | .negative_unaligned_scratch_offset_bug, |
| 2223 | .nsa_clause_bug, |
| 2224 | .nsa_encoding, |
| 2225 | .nsa_to_vmem_bug, |
| 2226 | .offset_3f_bug, |
| 2227 | .requires_cov6, |
| 2228 | .scalar_atomics, |
| 2229 | .scalar_flat_scratch_insts, |
| 2230 | .scalar_stores, |
| 2231 | .smem_to_vector_write_hazard, |
| 2232 | .vcmpx_exec_war_hazard, |
| 2233 | .vcmpx_permlane_hazard, |
| 2234 | .vmem_to_scalar_write_hazard, |
| 2235 | .xnack_support, |
| 2236 | }), |
| 2237 | }; |
| 2238 | pub const gfx10_3_generic: CpuModel = .{ |
| 2239 | .name = "gfx10_3_generic", |
| 2240 | .llvm_name = "gfx10-3-generic", |
| 2241 | .features = featureSet(&[_]Feature{ |
| 2242 | .back_off_barrier, |
| 2243 | .dl_insts, |
| 2244 | .dot10_insts, |
| 2245 | .dot1_insts, |
| 2246 | .dot2_insts, |
| 2247 | .dot5_insts, |
| 2248 | .dot6_insts, |
| 2249 | .dot7_insts, |
| 2250 | .gfx10, |
| 2251 | .gfx10_3_insts, |
| 2252 | .gfx10_a_encoding, |
| 2253 | .gfx10_b_encoding, |
| 2254 | .ldsbankcount32, |
| 2255 | .nsa_encoding, |
| 2256 | .requires_cov6, |
| 2257 | .shader_cycles_register, |
| 2258 | }), |
| 2259 | }; |
| 2260 | pub const gfx1100: CpuModel = .{ |
| 2261 | .name = "gfx1100", |
| 2262 | .llvm_name = "gfx1100", |
| 2263 | .features = featureSet(&[_]Feature{ |
| 2264 | .allocate1_5xvgprs, |
| 2265 | .architected_flat_scratch, |
| 2266 | .atomic_fadd_no_rtn_insts, |
| 2267 | .atomic_fadd_rtn_insts, |
| 2268 | .back_off_barrier, |
| 2269 | .d16_write_vgpr32, |
| 2270 | .dl_insts, |
| 2271 | .dot10_insts, |
| 2272 | .dot12_insts, |
| 2273 | .dot5_insts, |
| 2274 | .dot7_insts, |
| 2275 | .dot8_insts, |
| 2276 | .dot9_insts, |
| 2277 | .flat_atomic_fadd_f32_inst, |
| 2278 | .gfx11, |
| 2279 | .image_insts, |
| 2280 | .ldsbankcount32, |
| 2281 | .mad_intra_fwd_bug, |
| 2282 | .memory_atomic_fadd_f32_denormal_support, |
| 2283 | .msaa_load_dst_sel_bug, |
| 2284 | .nsa_encoding, |
| 2285 | .packed_tid, |
| 2286 | .partial_nsa_encoding, |
| 2287 | .priv_enabled_trap2_nop_bug, |
| 2288 | .real_true16, |
| 2289 | .shader_cycles_register, |
| 2290 | .user_sgpr_init16_bug, |
| 2291 | .valu_trans_use_hazard, |
| 2292 | .vcmpx_permlane_hazard, |
| 2293 | }), |
| 2294 | }; |
| 2295 | pub const gfx1101: CpuModel = .{ |
| 2296 | .name = "gfx1101", |
| 2297 | .llvm_name = "gfx1101", |
| 2298 | .features = featureSet(&[_]Feature{ |
| 2299 | .allocate1_5xvgprs, |
| 2300 | .architected_flat_scratch, |
| 2301 | .atomic_fadd_no_rtn_insts, |
| 2302 | .atomic_fadd_rtn_insts, |
| 2303 | .back_off_barrier, |
| 2304 | .d16_write_vgpr32, |
| 2305 | .dl_insts, |
| 2306 | .dot10_insts, |
| 2307 | .dot12_insts, |
| 2308 | .dot5_insts, |
| 2309 | .dot7_insts, |
| 2310 | .dot8_insts, |
| 2311 | .dot9_insts, |
| 2312 | .flat_atomic_fadd_f32_inst, |
| 2313 | .gfx11, |
| 2314 | .image_insts, |
| 2315 | .ldsbankcount32, |
| 2316 | .mad_intra_fwd_bug, |
| 2317 | .memory_atomic_fadd_f32_denormal_support, |
| 2318 | .msaa_load_dst_sel_bug, |
| 2319 | .nsa_encoding, |
| 2320 | .packed_tid, |
| 2321 | .partial_nsa_encoding, |
| 2322 | .priv_enabled_trap2_nop_bug, |
| 2323 | .real_true16, |
| 2324 | .shader_cycles_register, |
| 2325 | .valu_trans_use_hazard, |
| 2326 | .vcmpx_permlane_hazard, |
| 2327 | }), |
| 2328 | }; |
| 2329 | pub const gfx1102: CpuModel = .{ |
| 2330 | .name = "gfx1102", |
| 2331 | .llvm_name = "gfx1102", |
| 2332 | .features = featureSet(&[_]Feature{ |
| 2333 | .architected_flat_scratch, |
| 2334 | .atomic_fadd_no_rtn_insts, |
| 2335 | .atomic_fadd_rtn_insts, |
| 2336 | .back_off_barrier, |
| 2337 | .d16_write_vgpr32, |
| 2338 | .dl_insts, |
| 2339 | .dot10_insts, |
| 2340 | .dot12_insts, |
| 2341 | .dot5_insts, |
| 2342 | .dot7_insts, |
| 2343 | .dot8_insts, |
| 2344 | .dot9_insts, |
| 2345 | .flat_atomic_fadd_f32_inst, |
| 2346 | .gfx11, |
| 2347 | .image_insts, |
| 2348 | .ldsbankcount32, |
| 2349 | .mad_intra_fwd_bug, |
| 2350 | .memory_atomic_fadd_f32_denormal_support, |
| 2351 | .msaa_load_dst_sel_bug, |
| 2352 | .nsa_encoding, |
| 2353 | .packed_tid, |
| 2354 | .partial_nsa_encoding, |
| 2355 | .priv_enabled_trap2_nop_bug, |
| 2356 | .real_true16, |
| 2357 | .shader_cycles_register, |
| 2358 | .user_sgpr_init16_bug, |
| 2359 | .valu_trans_use_hazard, |
| 2360 | .vcmpx_permlane_hazard, |
| 2361 | }), |
| 2362 | }; |
| 2363 | pub const gfx1103: CpuModel = .{ |
| 2364 | .name = "gfx1103", |
| 2365 | .llvm_name = "gfx1103", |
| 2366 | .features = featureSet(&[_]Feature{ |
| 2367 | .architected_flat_scratch, |
| 2368 | .atomic_fadd_no_rtn_insts, |
| 2369 | .atomic_fadd_rtn_insts, |
| 2370 | .back_off_barrier, |
| 2371 | .d16_write_vgpr32, |
| 2372 | .dl_insts, |
| 2373 | .dot10_insts, |
| 2374 | .dot12_insts, |
| 2375 | .dot5_insts, |
| 2376 | .dot7_insts, |
| 2377 | .dot8_insts, |
| 2378 | .dot9_insts, |
| 2379 | .flat_atomic_fadd_f32_inst, |
| 2380 | .gfx11, |
| 2381 | .image_insts, |
| 2382 | .ldsbankcount32, |
| 2383 | .mad_intra_fwd_bug, |
| 2384 | .memory_atomic_fadd_f32_denormal_support, |
| 2385 | .msaa_load_dst_sel_bug, |
| 2386 | .nsa_encoding, |
| 2387 | .packed_tid, |
| 2388 | .partial_nsa_encoding, |
| 2389 | .priv_enabled_trap2_nop_bug, |
| 2390 | .real_true16, |
| 2391 | .shader_cycles_register, |
| 2392 | .valu_trans_use_hazard, |
| 2393 | .vcmpx_permlane_hazard, |
| 2394 | }), |
| 2395 | }; |
| 2396 | pub const gfx1150: CpuModel = .{ |
| 2397 | .name = "gfx1150", |
| 2398 | .llvm_name = "gfx1150", |
| 2399 | .features = featureSet(&[_]Feature{ |
| 2400 | .architected_flat_scratch, |
| 2401 | .atomic_fadd_no_rtn_insts, |
| 2402 | .atomic_fadd_rtn_insts, |
| 2403 | .back_off_barrier, |
| 2404 | .d16_write_vgpr32, |
| 2405 | .dl_insts, |
| 2406 | .dot10_insts, |
| 2407 | .dot12_insts, |
| 2408 | .dot5_insts, |
| 2409 | .dot7_insts, |
| 2410 | .dot8_insts, |
| 2411 | .dot9_insts, |
| 2412 | .dpp_src1_sgpr, |
| 2413 | .flat_atomic_fadd_f32_inst, |
| 2414 | .gfx11, |
| 2415 | .image_insts, |
| 2416 | .ldsbankcount32, |
| 2417 | .memory_atomic_fadd_f32_denormal_support, |
| 2418 | .nsa_encoding, |
| 2419 | .packed_tid, |
| 2420 | .partial_nsa_encoding, |
| 2421 | .point_sample_accel, |
| 2422 | .real_true16, |
| 2423 | .required_export_priority, |
| 2424 | .salu_float, |
| 2425 | .shader_cycles_register, |
| 2426 | .vcmpx_permlane_hazard, |
| 2427 | }), |
| 2428 | }; |
| 2429 | pub const gfx1151: CpuModel = .{ |
| 2430 | .name = "gfx1151", |
| 2431 | .llvm_name = "gfx1151", |
| 2432 | .features = featureSet(&[_]Feature{ |
| 2433 | .allocate1_5xvgprs, |
| 2434 | .architected_flat_scratch, |
| 2435 | .atomic_fadd_no_rtn_insts, |
| 2436 | .atomic_fadd_rtn_insts, |
| 2437 | .back_off_barrier, |
| 2438 | .d16_write_vgpr32, |
| 2439 | .dl_insts, |
| 2440 | .dot10_insts, |
| 2441 | .dot12_insts, |
| 2442 | .dot5_insts, |
| 2443 | .dot7_insts, |
| 2444 | .dot8_insts, |
| 2445 | .dot9_insts, |
| 2446 | .dpp_src1_sgpr, |
| 2447 | .flat_atomic_fadd_f32_inst, |
| 2448 | .gfx11, |
| 2449 | .image_insts, |
| 2450 | .ldsbankcount32, |
| 2451 | .memory_atomic_fadd_f32_denormal_support, |
| 2452 | .nsa_encoding, |
| 2453 | .packed_tid, |
| 2454 | .partial_nsa_encoding, |
| 2455 | .point_sample_accel, |
| 2456 | .real_true16, |
| 2457 | .required_export_priority, |
| 2458 | .salu_float, |
| 2459 | .shader_cycles_register, |
| 2460 | .vcmpx_permlane_hazard, |
| 2461 | }), |
| 2462 | }; |
| 2463 | pub const gfx1152: CpuModel = .{ |
| 2464 | .name = "gfx1152", |
| 2465 | .llvm_name = "gfx1152", |
| 2466 | .features = featureSet(&[_]Feature{ |
| 2467 | .architected_flat_scratch, |
| 2468 | .atomic_fadd_no_rtn_insts, |
| 2469 | .atomic_fadd_rtn_insts, |
| 2470 | .back_off_barrier, |
| 2471 | .d16_write_vgpr32, |
| 2472 | .dl_insts, |
| 2473 | .dot10_insts, |
| 2474 | .dot12_insts, |
| 2475 | .dot5_insts, |
| 2476 | .dot7_insts, |
| 2477 | .dot8_insts, |
| 2478 | .dot9_insts, |
| 2479 | .dpp_src1_sgpr, |
| 2480 | .flat_atomic_fadd_f32_inst, |
| 2481 | .gfx11, |
| 2482 | .image_insts, |
| 2483 | .ldsbankcount32, |
| 2484 | .memory_atomic_fadd_f32_denormal_support, |
| 2485 | .nsa_encoding, |
| 2486 | .packed_tid, |
| 2487 | .partial_nsa_encoding, |
| 2488 | .point_sample_accel, |
| 2489 | .real_true16, |
| 2490 | .required_export_priority, |
| 2491 | .salu_float, |
| 2492 | .shader_cycles_register, |
| 2493 | .vcmpx_permlane_hazard, |
| 2494 | }), |
| 2495 | }; |
| 2496 | pub const gfx1153: CpuModel = .{ |
| 2497 | .name = "gfx1153", |
| 2498 | .llvm_name = "gfx1153", |
| 2499 | .features = featureSet(&[_]Feature{ |
| 2500 | .architected_flat_scratch, |
| 2501 | .atomic_fadd_no_rtn_insts, |
| 2502 | .atomic_fadd_rtn_insts, |
| 2503 | .back_off_barrier, |
| 2504 | .d16_write_vgpr32, |
| 2505 | .dl_insts, |
| 2506 | .dot10_insts, |
| 2507 | .dot12_insts, |
| 2508 | .dot5_insts, |
| 2509 | .dot7_insts, |
| 2510 | .dot8_insts, |
| 2511 | .dot9_insts, |
| 2512 | .dpp_src1_sgpr, |
| 2513 | .flat_atomic_fadd_f32_inst, |
| 2514 | .gfx11, |
| 2515 | .image_insts, |
| 2516 | .ldsbankcount32, |
| 2517 | .memory_atomic_fadd_f32_denormal_support, |
| 2518 | .nsa_encoding, |
| 2519 | .packed_tid, |
| 2520 | .partial_nsa_encoding, |
| 2521 | .real_true16, |
| 2522 | .required_export_priority, |
| 2523 | .salu_float, |
| 2524 | .shader_cycles_register, |
| 2525 | .vcmpx_permlane_hazard, |
| 2526 | }), |
| 2527 | }; |
| 2528 | pub const gfx11_generic: CpuModel = .{ |
| 2529 | .name = "gfx11_generic", |
| 2530 | .llvm_name = "gfx11-generic", |
| 2531 | .features = featureSet(&[_]Feature{ |
| 2532 | .architected_flat_scratch, |
| 2533 | .atomic_fadd_no_rtn_insts, |
| 2534 | .atomic_fadd_rtn_insts, |
| 2535 | .back_off_barrier, |
| 2536 | .d16_write_vgpr32, |
| 2537 | .dl_insts, |
| 2538 | .dot10_insts, |
| 2539 | .dot12_insts, |
| 2540 | .dot5_insts, |
| 2541 | .dot7_insts, |
| 2542 | .dot8_insts, |
| 2543 | .dot9_insts, |
| 2544 | .flat_atomic_fadd_f32_inst, |
| 2545 | .gfx11, |
| 2546 | .image_insts, |
| 2547 | .ldsbankcount32, |
| 2548 | .mad_intra_fwd_bug, |
| 2549 | .memory_atomic_fadd_f32_denormal_support, |
| 2550 | .msaa_load_dst_sel_bug, |
| 2551 | .nsa_encoding, |
| 2552 | .packed_tid, |
| 2553 | .partial_nsa_encoding, |
| 2554 | .priv_enabled_trap2_nop_bug, |
| 2555 | .real_true16, |
| 2556 | .required_export_priority, |
| 2557 | .requires_cov6, |
| 2558 | .shader_cycles_register, |
| 2559 | .user_sgpr_init16_bug, |
| 2560 | .valu_trans_use_hazard, |
| 2561 | .vcmpx_permlane_hazard, |
| 2562 | }), |
| 2563 | }; |
| 2564 | pub const gfx1200: CpuModel = .{ |
| 2565 | .name = "gfx1200", |
| 2566 | .llvm_name = "gfx1200", |
| 2567 | .features = featureSet(&[_]Feature{ |
| 2568 | .addressablelocalmemorysize65536, |
| 2569 | .allocate1_5xvgprs, |
| 2570 | .architected_flat_scratch, |
| 2571 | .architected_sgprs, |
| 2572 | .atomic_buffer_global_pk_add_f16_insts, |
| 2573 | .atomic_buffer_pk_add_bf16_inst, |
| 2574 | .atomic_ds_pk_add_16_insts, |
| 2575 | .atomic_fadd_no_rtn_insts, |
| 2576 | .atomic_fadd_rtn_insts, |
| 2577 | .atomic_flat_pk_add_16_insts, |
| 2578 | .atomic_global_pk_add_bf16_inst, |
| 2579 | .back_off_barrier, |
| 2580 | .bvh_dual_bvh_8_insts, |
| 2581 | .cube_insts, |
| 2582 | .cvt_norm_insts, |
| 2583 | .cvt_pknorm_vop2_insts, |
| 2584 | .cvt_pknorm_vop3_insts, |
| 2585 | .d16_write_vgpr32, |
| 2586 | .dl_insts, |
| 2587 | .dot10_insts, |
| 2588 | .dot11_insts, |
| 2589 | .dot12_insts, |
| 2590 | .dot7_insts, |
| 2591 | .dot8_insts, |
| 2592 | .dot9_insts, |
| 2593 | .dpp_src1_sgpr, |
| 2594 | .extended_image_insts, |
| 2595 | .flat_atomic_fadd_f32_inst, |
| 2596 | .fp8_conversion_insts, |
| 2597 | .gfx12, |
| 2598 | .image_insts, |
| 2599 | .ldsbankcount32, |
| 2600 | .lerp_inst, |
| 2601 | .memory_atomic_fadd_f32_denormal_support, |
| 2602 | .nsa_encoding, |
| 2603 | .packed_tid, |
| 2604 | .partial_nsa_encoding, |
| 2605 | .pseudo_scalar_trans, |
| 2606 | .qsad_insts, |
| 2607 | .restricted_soffset, |
| 2608 | .sad_insts, |
| 2609 | .salu_float, |
| 2610 | .scalar_dwordx3_loads, |
| 2611 | .shader_cycles_hi_lo_registers, |
| 2612 | .vcmpx_permlane_hazard, |
| 2613 | .waits_before_system_scope_stores, |
| 2614 | }), |
| 2615 | }; |
| 2616 | pub const gfx1201: CpuModel = .{ |
| 2617 | .name = "gfx1201", |
| 2618 | .llvm_name = "gfx1201", |
| 2619 | .features = featureSet(&[_]Feature{ |
| 2620 | .addressablelocalmemorysize65536, |
| 2621 | .allocate1_5xvgprs, |
| 2622 | .architected_flat_scratch, |
| 2623 | .architected_sgprs, |
| 2624 | .atomic_buffer_global_pk_add_f16_insts, |
| 2625 | .atomic_buffer_pk_add_bf16_inst, |
| 2626 | .atomic_ds_pk_add_16_insts, |
| 2627 | .atomic_fadd_no_rtn_insts, |
| 2628 | .atomic_fadd_rtn_insts, |
| 2629 | .atomic_flat_pk_add_16_insts, |
| 2630 | .atomic_global_pk_add_bf16_inst, |
| 2631 | .back_off_barrier, |
| 2632 | .bvh_dual_bvh_8_insts, |
| 2633 | .cube_insts, |
| 2634 | .cvt_norm_insts, |
| 2635 | .cvt_pknorm_vop2_insts, |
| 2636 | .cvt_pknorm_vop3_insts, |
| 2637 | .d16_write_vgpr32, |
| 2638 | .dl_insts, |
| 2639 | .dot10_insts, |
| 2640 | .dot11_insts, |
| 2641 | .dot12_insts, |
| 2642 | .dot7_insts, |
| 2643 | .dot8_insts, |
| 2644 | .dot9_insts, |
| 2645 | .dpp_src1_sgpr, |
| 2646 | .extended_image_insts, |
| 2647 | .flat_atomic_fadd_f32_inst, |
| 2648 | .fp8_conversion_insts, |
| 2649 | .gfx12, |
| 2650 | .image_insts, |
| 2651 | .ldsbankcount32, |
| 2652 | .lerp_inst, |
| 2653 | .memory_atomic_fadd_f32_denormal_support, |
| 2654 | .nsa_encoding, |
| 2655 | .packed_tid, |
| 2656 | .partial_nsa_encoding, |
| 2657 | .pseudo_scalar_trans, |
| 2658 | .qsad_insts, |
| 2659 | .restricted_soffset, |
| 2660 | .sad_insts, |
| 2661 | .salu_float, |
| 2662 | .scalar_dwordx3_loads, |
| 2663 | .shader_cycles_hi_lo_registers, |
| 2664 | .vcmpx_permlane_hazard, |
| 2665 | .waits_before_system_scope_stores, |
| 2666 | }), |
| 2667 | }; |
| 2668 | pub const gfx1250: CpuModel = .{ |
| 2669 | .name = "gfx1250", |
| 2670 | .llvm_name = "gfx1250", |
| 2671 | .features = featureSet(&[_]Feature{ |
| 2672 | .@"1024_addressable_vgprs", |
| 2673 | .@"45_bit_num_records_buffer_resource", |
| 2674 | .@"64_bit_literals", |
| 2675 | .add_min_max_insts, |
| 2676 | .add_sub_u64_insts, |
| 2677 | .addressablelocalmemorysize327680, |
| 2678 | .architected_flat_scratch, |
| 2679 | .architected_sgprs, |
| 2680 | .ashr_pk_insts, |
| 2681 | .atomic_buffer_global_pk_add_f16_insts, |
| 2682 | .atomic_buffer_pk_add_bf16_inst, |
| 2683 | .atomic_ds_pk_add_16_insts, |
| 2684 | .atomic_fadd_no_rtn_insts, |
| 2685 | .atomic_fadd_rtn_insts, |
| 2686 | .atomic_flat_pk_add_16_insts, |
| 2687 | .atomic_fmin_fmax_flat_f64, |
| 2688 | .atomic_fmin_fmax_global_f64, |
| 2689 | .atomic_global_pk_add_bf16_inst, |
| 2690 | .bf16_cvt_insts, |
| 2691 | .bf16_pk_insts, |
| 2692 | .bf16_trans_insts, |
| 2693 | .bitop3_insts, |
| 2694 | .clusters, |
| 2695 | .cube_insts, |
| 2696 | .cumode, |
| 2697 | .cvt_norm_insts, |
| 2698 | .cvt_pk_f16_f32_inst, |
| 2699 | .cvt_pknorm_vop2_insts, |
| 2700 | .cvt_pknorm_vop3_insts, |
| 2701 | .d16_write_vgpr32, |
| 2702 | .dl_insts, |
| 2703 | .dot7_insts, |
| 2704 | .dot8_insts, |
| 2705 | .dpp_src1_sgpr, |
| 2706 | .emulated_system_scope_atomics, |
| 2707 | .flat_atomic_fadd_f32_inst, |
| 2708 | .flat_buffer_global_fadd_f64_inst, |
| 2709 | .flat_gvs_mode, |
| 2710 | .fma_mix_bf16_insts, |
| 2711 | .fmacf64_inst, |
| 2712 | .fp8_conversion_insts, |
| 2713 | .fp8e5m3_insts, |
| 2714 | .gfx12, |
| 2715 | .gfx1250_insts, |
| 2716 | .globally_addressable_scratch, |
| 2717 | .kernarg_preload, |
| 2718 | .lds_barrier_arrive_atomic, |
| 2719 | .ldsbankcount32, |
| 2720 | .lerp_inst, |
| 2721 | .lshl_add_u64_inst, |
| 2722 | .mad_u32_inst, |
| 2723 | .max_hard_clause_length_63, |
| 2724 | .mcast_load_insts, |
| 2725 | .memory_atomic_fadd_f32_denormal_support, |
| 2726 | .min3_max3_pkf16, |
| 2727 | .minimum3_maximum3_pkf16, |
| 2728 | .packed_fp32_ops, |
| 2729 | .packed_tid, |
| 2730 | .permlane16_swap, |
| 2731 | .pk_add_min_max_insts, |
| 2732 | .prng_inst, |
| 2733 | .pseudo_scalar_trans, |
| 2734 | .qsad_insts, |
| 2735 | .restricted_soffset, |
| 2736 | .s_wakeup_barrier_inst, |
| 2737 | .sad_insts, |
| 2738 | .salu_float, |
| 2739 | .scalar_dwordx3_loads, |
| 2740 | .setprio_inc_wg_inst, |
| 2741 | .setreg_vgpr_msb_fixup, |
| 2742 | .shader_cycles_hi_lo_registers, |
| 2743 | .sramecc_support, |
| 2744 | .tanh_insts, |
| 2745 | .tensor_cvt_lut_insts, |
| 2746 | .transpose_load_f4f6_insts, |
| 2747 | .vcmpx_permlane_hazard, |
| 2748 | .vgpr_align2, |
| 2749 | .vmem_pref_insts, |
| 2750 | .wait_xcnt, |
| 2751 | .wavefrontsize32, |
| 2752 | .xnack, |
| 2753 | .xnack_support, |
| 2754 | }), |
| 2755 | }; |
| 2756 | pub const gfx1251: CpuModel = .{ |
| 2757 | .name = "gfx1251", |
| 2758 | .llvm_name = "gfx1251", |
| 2759 | .features = featureSet(&[_]Feature{ |
| 2760 | .@"1024_addressable_vgprs", |
| 2761 | .@"45_bit_num_records_buffer_resource", |
| 2762 | .@"64_bit_literals", |
| 2763 | .add_min_max_insts, |
| 2764 | .add_sub_u64_insts, |
| 2765 | .addressablelocalmemorysize327680, |
| 2766 | .architected_flat_scratch, |
| 2767 | .architected_sgprs, |
| 2768 | .ashr_pk_insts, |
| 2769 | .atomic_buffer_global_pk_add_f16_insts, |
| 2770 | .atomic_buffer_pk_add_bf16_inst, |
| 2771 | .atomic_ds_pk_add_16_insts, |
| 2772 | .atomic_fadd_no_rtn_insts, |
| 2773 | .atomic_fadd_rtn_insts, |
| 2774 | .atomic_flat_pk_add_16_insts, |
| 2775 | .atomic_fmin_fmax_flat_f64, |
| 2776 | .atomic_fmin_fmax_global_f64, |
| 2777 | .atomic_global_pk_add_bf16_inst, |
| 2778 | .bf16_cvt_insts, |
| 2779 | .bf16_pk_insts, |
| 2780 | .bf16_trans_insts, |
| 2781 | .bitop3_insts, |
| 2782 | .clusters, |
| 2783 | .cube_insts, |
| 2784 | .cumode, |
| 2785 | .cvt_norm_insts, |
| 2786 | .cvt_pk_f16_f32_inst, |
| 2787 | .cvt_pknorm_vop2_insts, |
| 2788 | .cvt_pknorm_vop3_insts, |
| 2789 | .d16_write_vgpr32, |
| 2790 | .dl_insts, |
| 2791 | .dot7_insts, |
| 2792 | .dot8_insts, |
| 2793 | .dpp_64bit, |
| 2794 | .dpp_src1_sgpr, |
| 2795 | .emulated_system_scope_atomics, |
| 2796 | .flat_atomic_fadd_f32_inst, |
| 2797 | .flat_buffer_global_fadd_f64_inst, |
| 2798 | .flat_gvs_mode, |
| 2799 | .fma_mix_bf16_insts, |
| 2800 | .fmacf64_inst, |
| 2801 | .fp8_conversion_insts, |
| 2802 | .fp8e5m3_insts, |
| 2803 | .gfx12, |
| 2804 | .gfx1250_insts, |
| 2805 | .globally_addressable_scratch, |
| 2806 | .kernarg_preload, |
| 2807 | .lds_barrier_arrive_atomic, |
| 2808 | .ldsbankcount32, |
| 2809 | .lerp_inst, |
| 2810 | .lshl_add_u64_inst, |
| 2811 | .mad_u32_inst, |
| 2812 | .max_hard_clause_length_63, |
| 2813 | .mcast_load_insts, |
| 2814 | .memory_atomic_fadd_f32_denormal_support, |
| 2815 | .min3_max3_pkf16, |
| 2816 | .minimum3_maximum3_pkf16, |
| 2817 | .packed_fp32_ops, |
| 2818 | .packed_tid, |
| 2819 | .permlane16_swap, |
| 2820 | .pk_add_min_max_insts, |
| 2821 | .prng_inst, |
| 2822 | .pseudo_scalar_trans, |
| 2823 | .qsad_insts, |
| 2824 | .restricted_soffset, |
| 2825 | .s_wakeup_barrier_inst, |
| 2826 | .sad_insts, |
| 2827 | .salu_float, |
| 2828 | .scalar_dwordx3_loads, |
| 2829 | .setprio_inc_wg_inst, |
| 2830 | .shader_cycles_hi_lo_registers, |
| 2831 | .sramecc_support, |
| 2832 | .tanh_insts, |
| 2833 | .tensor_cvt_lut_insts, |
| 2834 | .transpose_load_f4f6_insts, |
| 2835 | .vcmpx_permlane_hazard, |
| 2836 | .vgpr_align2, |
| 2837 | .vmem_pref_insts, |
| 2838 | .wait_xcnt, |
| 2839 | .wavefrontsize32, |
| 2840 | .xnack, |
| 2841 | .xnack_support, |
| 2842 | }), |
| 2843 | }; |
| 2844 | pub const gfx12_generic: CpuModel = .{ |
| 2845 | .name = "gfx12_generic", |
| 2846 | .llvm_name = "gfx12-generic", |
| 2847 | .features = featureSet(&[_]Feature{ |
| 2848 | .addressablelocalmemorysize65536, |
| 2849 | .allocate1_5xvgprs, |
| 2850 | .architected_flat_scratch, |
| 2851 | .architected_sgprs, |
| 2852 | .atomic_buffer_global_pk_add_f16_insts, |
| 2853 | .atomic_buffer_pk_add_bf16_inst, |
| 2854 | .atomic_ds_pk_add_16_insts, |
| 2855 | .atomic_fadd_no_rtn_insts, |
| 2856 | .atomic_fadd_rtn_insts, |
| 2857 | .atomic_flat_pk_add_16_insts, |
| 2858 | .atomic_global_pk_add_bf16_inst, |
| 2859 | .back_off_barrier, |
| 2860 | .bvh_dual_bvh_8_insts, |
| 2861 | .cube_insts, |
| 2862 | .cvt_norm_insts, |
| 2863 | .cvt_pknorm_vop2_insts, |
| 2864 | .cvt_pknorm_vop3_insts, |
| 2865 | .d16_write_vgpr32, |
| 2866 | .dl_insts, |
| 2867 | .dot10_insts, |
| 2868 | .dot11_insts, |
| 2869 | .dot12_insts, |
| 2870 | .dot7_insts, |
| 2871 | .dot8_insts, |
| 2872 | .dot9_insts, |
| 2873 | .dpp_src1_sgpr, |
| 2874 | .extended_image_insts, |
| 2875 | .flat_atomic_fadd_f32_inst, |
| 2876 | .fp8_conversion_insts, |
| 2877 | .gfx12, |
| 2878 | .image_insts, |
| 2879 | .ldsbankcount32, |
| 2880 | .lerp_inst, |
| 2881 | .memory_atomic_fadd_f32_denormal_support, |
| 2882 | .nsa_encoding, |
| 2883 | .packed_tid, |
| 2884 | .partial_nsa_encoding, |
| 2885 | .pseudo_scalar_trans, |
| 2886 | .qsad_insts, |
| 2887 | .requires_cov6, |
| 2888 | .restricted_soffset, |
| 2889 | .sad_insts, |
| 2890 | .salu_float, |
| 2891 | .scalar_dwordx3_loads, |
| 2892 | .shader_cycles_hi_lo_registers, |
| 2893 | .vcmpx_permlane_hazard, |
| 2894 | .waits_before_system_scope_stores, |
| 2895 | }), |
| 2896 | }; |
| 2897 | pub const gfx600: CpuModel = .{ |
| 2898 | .name = "gfx600", |
| 2899 | .llvm_name = "gfx600", |
| 2900 | .features = featureSet(&[_]Feature{ |
| 2901 | .fast_fmaf, |
| 2902 | .half_rate_64_ops, |
| 2903 | .southern_islands, |
| 2904 | }), |
| 2905 | }; |
| 2906 | pub const gfx601: CpuModel = .{ |
| 2907 | .name = "gfx601", |
| 2908 | .llvm_name = "gfx601", |
| 2909 | .features = featureSet(&[_]Feature{ |
| 2910 | .southern_islands, |
| 2911 | }), |
| 2912 | }; |
| 2913 | pub const gfx602: CpuModel = .{ |
| 2914 | .name = "gfx602", |
| 2915 | .llvm_name = "gfx602", |
| 2916 | .features = featureSet(&[_]Feature{ |
| 2917 | .southern_islands, |
| 2918 | }), |
| 2919 | }; |
| 2920 | pub const gfx700: CpuModel = .{ |
| 2921 | .name = "gfx700", |
| 2922 | .llvm_name = "gfx700", |
| 2923 | .features = featureSet(&[_]Feature{ |
| 2924 | .ldsbankcount32, |
| 2925 | .sea_islands, |
| 2926 | }), |
| 2927 | }; |
| 2928 | pub const gfx701: CpuModel = .{ |
| 2929 | .name = "gfx701", |
| 2930 | .llvm_name = "gfx701", |
| 2931 | .features = featureSet(&[_]Feature{ |
| 2932 | .fast_fmaf, |
| 2933 | .half_rate_64_ops, |
| 2934 | .ldsbankcount32, |
| 2935 | .sea_islands, |
| 2936 | }), |
| 2937 | }; |
| 2938 | pub const gfx702: CpuModel = .{ |
| 2939 | .name = "gfx702", |
| 2940 | .llvm_name = "gfx702", |
| 2941 | .features = featureSet(&[_]Feature{ |
| 2942 | .fast_fmaf, |
| 2943 | .ldsbankcount16, |
| 2944 | .sea_islands, |
| 2945 | }), |
| 2946 | }; |
| 2947 | pub const gfx703: CpuModel = .{ |
| 2948 | .name = "gfx703", |
| 2949 | .llvm_name = "gfx703", |
| 2950 | .features = featureSet(&[_]Feature{ |
| 2951 | .ldsbankcount16, |
| 2952 | .sea_islands, |
| 2953 | }), |
| 2954 | }; |
| 2955 | pub const gfx704: CpuModel = .{ |
| 2956 | .name = "gfx704", |
| 2957 | .llvm_name = "gfx704", |
| 2958 | .features = featureSet(&[_]Feature{ |
| 2959 | .ldsbankcount32, |
| 2960 | .sea_islands, |
| 2961 | }), |
| 2962 | }; |
| 2963 | pub const gfx705: CpuModel = .{ |
| 2964 | .name = "gfx705", |
| 2965 | .llvm_name = "gfx705", |
| 2966 | .features = featureSet(&[_]Feature{ |
| 2967 | .ldsbankcount16, |
| 2968 | .sea_islands, |
| 2969 | }), |
| 2970 | }; |
| 2971 | pub const gfx801: CpuModel = .{ |
| 2972 | .name = "gfx801", |
| 2973 | .llvm_name = "gfx801", |
| 2974 | .features = featureSet(&[_]Feature{ |
| 2975 | .fast_fmaf, |
| 2976 | .half_rate_64_ops, |
| 2977 | .ldsbankcount32, |
| 2978 | .unpacked_d16_vmem, |
| 2979 | .volcanic_islands, |
| 2980 | .xnack_support, |
| 2981 | }), |
| 2982 | }; |
| 2983 | pub const gfx802: CpuModel = .{ |
| 2984 | .name = "gfx802", |
| 2985 | .llvm_name = "gfx802", |
| 2986 | .features = featureSet(&[_]Feature{ |
| 2987 | .ldsbankcount32, |
| 2988 | .sgpr_init_bug, |
| 2989 | .unpacked_d16_vmem, |
| 2990 | .volcanic_islands, |
| 2991 | }), |
| 2992 | }; |
| 2993 | pub const gfx803: CpuModel = .{ |
| 2994 | .name = "gfx803", |
| 2995 | .llvm_name = "gfx803", |
| 2996 | .features = featureSet(&[_]Feature{ |
| 2997 | .ldsbankcount32, |
| 2998 | .unpacked_d16_vmem, |
| 2999 | .volcanic_islands, |
| 3000 | }), |
| 3001 | }; |
| 3002 | pub const gfx805: CpuModel = .{ |
| 3003 | .name = "gfx805", |
| 3004 | .llvm_name = "gfx805", |
| 3005 | .features = featureSet(&[_]Feature{ |
| 3006 | .ldsbankcount32, |
| 3007 | .sgpr_init_bug, |
| 3008 | .unpacked_d16_vmem, |
| 3009 | .volcanic_islands, |
| 3010 | }), |
| 3011 | }; |
| 3012 | pub const gfx810: CpuModel = .{ |
| 3013 | .name = "gfx810", |
| 3014 | .llvm_name = "gfx810", |
| 3015 | .features = featureSet(&[_]Feature{ |
| 3016 | .image_gather4_d16_bug, |
| 3017 | .image_store_d16_bug, |
| 3018 | .ldsbankcount16, |
| 3019 | .volcanic_islands, |
| 3020 | .xnack_support, |
| 3021 | }), |
| 3022 | }; |
| 3023 | pub const gfx900: CpuModel = .{ |
| 3024 | .name = "gfx900", |
| 3025 | .llvm_name = "gfx900", |
| 3026 | .features = featureSet(&[_]Feature{ |
| 3027 | .addressablelocalmemorysize65536, |
| 3028 | .ds_src2_insts, |
| 3029 | .extended_image_insts, |
| 3030 | .gds, |
| 3031 | .gfx9, |
| 3032 | .image_gather4_d16_bug, |
| 3033 | .image_insts, |
| 3034 | .ldsbankcount32, |
| 3035 | .mad_mac_f32_insts, |
| 3036 | .mad_mix_insts, |
| 3037 | }), |
| 3038 | }; |
| 3039 | pub const gfx902: CpuModel = .{ |
| 3040 | .name = "gfx902", |
| 3041 | .llvm_name = "gfx902", |
| 3042 | .features = featureSet(&[_]Feature{ |
| 3043 | .addressablelocalmemorysize65536, |
| 3044 | .ds_src2_insts, |
| 3045 | .extended_image_insts, |
| 3046 | .gds, |
| 3047 | .gfx9, |
| 3048 | .image_gather4_d16_bug, |
| 3049 | .image_insts, |
| 3050 | .ldsbankcount32, |
| 3051 | .mad_mac_f32_insts, |
| 3052 | .mad_mix_insts, |
| 3053 | }), |
| 3054 | }; |
| 3055 | pub const gfx904: CpuModel = .{ |
| 3056 | .name = "gfx904", |
| 3057 | .llvm_name = "gfx904", |
| 3058 | .features = featureSet(&[_]Feature{ |
| 3059 | .addressablelocalmemorysize65536, |
| 3060 | .ds_src2_insts, |
| 3061 | .extended_image_insts, |
| 3062 | .fma_mix_insts, |
| 3063 | .gds, |
| 3064 | .gfx9, |
| 3065 | .image_gather4_d16_bug, |
| 3066 | .image_insts, |
| 3067 | .ldsbankcount32, |
| 3068 | .mad_mac_f32_insts, |
| 3069 | }), |
| 3070 | }; |
| 3071 | pub const gfx906: CpuModel = .{ |
| 3072 | .name = "gfx906", |
| 3073 | .llvm_name = "gfx906", |
| 3074 | .features = featureSet(&[_]Feature{ |
| 3075 | .addressablelocalmemorysize65536, |
| 3076 | .dl_insts, |
| 3077 | .dot10_insts, |
| 3078 | .dot1_insts, |
| 3079 | .dot2_insts, |
| 3080 | .dot7_insts, |
| 3081 | .ds_src2_insts, |
| 3082 | .extended_image_insts, |
| 3083 | .fma_mix_insts, |
| 3084 | .gds, |
| 3085 | .gfx9, |
| 3086 | .half_rate_64_ops, |
| 3087 | .image_gather4_d16_bug, |
| 3088 | .image_insts, |
| 3089 | .ldsbankcount32, |
| 3090 | .mad_mac_f32_insts, |
| 3091 | .sramecc_support, |
| 3092 | }), |
| 3093 | }; |
| 3094 | pub const gfx908: CpuModel = .{ |
| 3095 | .name = "gfx908", |
| 3096 | .llvm_name = "gfx908", |
| 3097 | .features = featureSet(&[_]Feature{ |
| 3098 | .addressablelocalmemorysize65536, |
| 3099 | .atomic_buffer_global_pk_add_f16_no_rtn_insts, |
| 3100 | .atomic_fadd_no_rtn_insts, |
| 3101 | .dl_insts, |
| 3102 | .dot10_insts, |
| 3103 | .dot1_insts, |
| 3104 | .dot2_insts, |
| 3105 | .dot3_insts, |
| 3106 | .dot4_insts, |
| 3107 | .dot5_insts, |
| 3108 | .dot6_insts, |
| 3109 | .dot7_insts, |
| 3110 | .ds_src2_insts, |
| 3111 | .extended_image_insts, |
| 3112 | .fma_mix_insts, |
| 3113 | .gds, |
| 3114 | .gfx9, |
| 3115 | .half_rate_64_ops, |
| 3116 | .image_gather4_d16_bug, |
| 3117 | .image_insts, |
| 3118 | .ldsbankcount32, |
| 3119 | .mad_mac_f32_insts, |
| 3120 | .mai_insts, |
| 3121 | .mfma_inline_literal_bug, |
| 3122 | .pk_fmac_f16_inst, |
| 3123 | .sramecc_support, |
| 3124 | }), |
| 3125 | }; |
| 3126 | pub const gfx909: CpuModel = .{ |
| 3127 | .name = "gfx909", |
| 3128 | .llvm_name = "gfx909", |
| 3129 | .features = featureSet(&[_]Feature{ |
| 3130 | .addressablelocalmemorysize65536, |
| 3131 | .ds_src2_insts, |
| 3132 | .extended_image_insts, |
| 3133 | .gds, |
| 3134 | .gfx9, |
| 3135 | .image_gather4_d16_bug, |
| 3136 | .image_insts, |
| 3137 | .ldsbankcount32, |
| 3138 | .mad_mac_f32_insts, |
| 3139 | .mad_mix_insts, |
| 3140 | }), |
| 3141 | }; |
| 3142 | pub const gfx90a: CpuModel = .{ |
| 3143 | .name = "gfx90a", |
| 3144 | .llvm_name = "gfx90a", |
| 3145 | .features = featureSet(&[_]Feature{ |
| 3146 | .addressablelocalmemorysize65536, |
| 3147 | .atomic_buffer_global_pk_add_f16_insts, |
| 3148 | .atomic_fadd_no_rtn_insts, |
| 3149 | .atomic_fadd_rtn_insts, |
| 3150 | .atomic_fmin_fmax_flat_f64, |
| 3151 | .atomic_fmin_fmax_global_f64, |
| 3152 | .back_off_barrier, |
| 3153 | .dl_insts, |
| 3154 | .dot10_insts, |
| 3155 | .dot1_insts, |
| 3156 | .dot2_insts, |
| 3157 | .dot3_insts, |
| 3158 | .dot4_insts, |
| 3159 | .dot5_insts, |
| 3160 | .dot6_insts, |
| 3161 | .dot7_insts, |
| 3162 | .dpp_64bit, |
| 3163 | .flat_buffer_global_fadd_f64_inst, |
| 3164 | .fma_mix_insts, |
| 3165 | .fmacf64_inst, |
| 3166 | .full_rate_64_ops, |
| 3167 | .gfx9, |
| 3168 | .gfx90a_insts, |
| 3169 | .image_insts, |
| 3170 | .kernarg_preload, |
| 3171 | .ldsbankcount32, |
| 3172 | .mad_mac_f32_insts, |
| 3173 | .mai_insts, |
| 3174 | .packed_fp32_ops, |
| 3175 | .packed_tid, |
| 3176 | .pk_fmac_f16_inst, |
| 3177 | .sramecc_support, |
| 3178 | .vgpr_align2, |
| 3179 | }), |
| 3180 | }; |
| 3181 | pub const gfx90c: CpuModel = .{ |
| 3182 | .name = "gfx90c", |
| 3183 | .llvm_name = "gfx90c", |
| 3184 | .features = featureSet(&[_]Feature{ |
| 3185 | .addressablelocalmemorysize65536, |
| 3186 | .ds_src2_insts, |
| 3187 | .extended_image_insts, |
| 3188 | .gds, |
| 3189 | .gfx9, |
| 3190 | .image_gather4_d16_bug, |
| 3191 | .image_insts, |
| 3192 | .ldsbankcount32, |
| 3193 | .mad_mac_f32_insts, |
| 3194 | .mad_mix_insts, |
| 3195 | }), |
| 3196 | }; |
| 3197 | pub const gfx942: CpuModel = .{ |
| 3198 | .name = "gfx942", |
| 3199 | .llvm_name = "gfx942", |
| 3200 | .features = featureSet(&[_]Feature{ |
| 3201 | .addressablelocalmemorysize65536, |
| 3202 | .agent_scope_fine_grained_remote_memory_atomics, |
| 3203 | .architected_flat_scratch, |
| 3204 | .atomic_buffer_global_pk_add_f16_insts, |
| 3205 | .atomic_ds_pk_add_16_insts, |
| 3206 | .atomic_fadd_no_rtn_insts, |
| 3207 | .atomic_fadd_rtn_insts, |
| 3208 | .atomic_flat_pk_add_16_insts, |
| 3209 | .atomic_fmin_fmax_flat_f64, |
| 3210 | .atomic_fmin_fmax_global_f64, |
| 3211 | .atomic_global_pk_add_bf16_inst, |
| 3212 | .back_off_barrier, |
| 3213 | .cvt_fp8_vop1_bug, |
| 3214 | .dl_insts, |
| 3215 | .dot10_insts, |
| 3216 | .dot1_insts, |
| 3217 | .dot2_insts, |
| 3218 | .dot3_insts, |
| 3219 | .dot4_insts, |
| 3220 | .dot5_insts, |
| 3221 | .dot6_insts, |
| 3222 | .dot7_insts, |
| 3223 | .dpp_64bit, |
| 3224 | .flat_atomic_fadd_f32_inst, |
| 3225 | .flat_buffer_global_fadd_f64_inst, |
| 3226 | .fma_mix_insts, |
| 3227 | .fmacf64_inst, |
| 3228 | .fp8_insts, |
| 3229 | .full_rate_64_ops, |
| 3230 | .gfx9, |
| 3231 | .gfx90a_insts, |
| 3232 | .gfx940_insts, |
| 3233 | .kernarg_preload, |
| 3234 | .ldsbankcount32, |
| 3235 | .lshl_add_u64_inst, |
| 3236 | .mai_insts, |
| 3237 | .memory_atomic_fadd_f32_denormal_support, |
| 3238 | .packed_fp32_ops, |
| 3239 | .packed_tid, |
| 3240 | .pk_fmac_f16_inst, |
| 3241 | .sramecc_support, |
| 3242 | .vgpr_align2, |
| 3243 | .xf32_insts, |
| 3244 | }), |
| 3245 | }; |
| 3246 | pub const gfx950: CpuModel = .{ |
| 3247 | .name = "gfx950", |
| 3248 | .llvm_name = "gfx950", |
| 3249 | .features = featureSet(&[_]Feature{ |
| 3250 | .addressablelocalmemorysize163840, |
| 3251 | .agent_scope_fine_grained_remote_memory_atomics, |
| 3252 | .architected_flat_scratch, |
| 3253 | .atomic_buffer_global_pk_add_f16_insts, |
| 3254 | .atomic_buffer_pk_add_bf16_inst, |
| 3255 | .atomic_ds_pk_add_16_insts, |
| 3256 | .atomic_fadd_no_rtn_insts, |
| 3257 | .atomic_fadd_rtn_insts, |
| 3258 | .atomic_flat_pk_add_16_insts, |
| 3259 | .atomic_fmin_fmax_flat_f64, |
| 3260 | .atomic_fmin_fmax_global_f64, |
| 3261 | .atomic_global_pk_add_bf16_inst, |
| 3262 | .back_off_barrier, |
| 3263 | .bf16_cvt_insts, |
| 3264 | .bitop3_insts, |
| 3265 | .dl_insts, |
| 3266 | .dot10_insts, |
| 3267 | .dot12_insts, |
| 3268 | .dot13_insts, |
| 3269 | .dot1_insts, |
| 3270 | .dot2_insts, |
| 3271 | .dot3_insts, |
| 3272 | .dot4_insts, |
| 3273 | .dot5_insts, |
| 3274 | .dot6_insts, |
| 3275 | .dot7_insts, |
| 3276 | .dpp_64bit, |
| 3277 | .flat_atomic_fadd_f32_inst, |
| 3278 | .flat_buffer_global_fadd_f64_inst, |
| 3279 | .fma_mix_insts, |
| 3280 | .fmacf64_inst, |
| 3281 | .fp8_conversion_insts, |
| 3282 | .fp8_insts, |
| 3283 | .full_rate_64_ops, |
| 3284 | .gfx9, |
| 3285 | .gfx90a_insts, |
| 3286 | .gfx940_insts, |
| 3287 | .gfx950_insts, |
| 3288 | .kernarg_preload, |
| 3289 | .ldsbankcount32, |
| 3290 | .lshl_add_u64_inst, |
| 3291 | .mai_insts, |
| 3292 | .memory_atomic_fadd_f32_denormal_support, |
| 3293 | .packed_fp32_ops, |
| 3294 | .packed_tid, |
| 3295 | .pk_fmac_f16_inst, |
| 3296 | .prng_inst, |
| 3297 | .sramecc_support, |
| 3298 | .vgpr_align2, |
| 3299 | }), |
| 3300 | }; |
| 3301 | pub const gfx9_4_generic: CpuModel = .{ |
| 3302 | .name = "gfx9_4_generic", |
| 3303 | .llvm_name = "gfx9-4-generic", |
| 3304 | .features = featureSet(&[_]Feature{ |
| 3305 | .addressablelocalmemorysize65536, |
| 3306 | .agent_scope_fine_grained_remote_memory_atomics, |
| 3307 | .architected_flat_scratch, |
| 3308 | .atomic_buffer_global_pk_add_f16_insts, |
| 3309 | .atomic_ds_pk_add_16_insts, |
| 3310 | .atomic_fadd_no_rtn_insts, |
| 3311 | .atomic_fadd_rtn_insts, |
| 3312 | .atomic_flat_pk_add_16_insts, |
| 3313 | .atomic_fmin_fmax_flat_f64, |
| 3314 | .atomic_fmin_fmax_global_f64, |
| 3315 | .atomic_global_pk_add_bf16_inst, |
| 3316 | .back_off_barrier, |
| 3317 | .dl_insts, |
| 3318 | .dot10_insts, |
| 3319 | .dot1_insts, |
| 3320 | .dot2_insts, |
| 3321 | .dot3_insts, |
| 3322 | .dot4_insts, |
| 3323 | .dot5_insts, |
| 3324 | .dot6_insts, |
| 3325 | .dot7_insts, |
| 3326 | .dpp_64bit, |
| 3327 | .flat_atomic_fadd_f32_inst, |
| 3328 | .flat_buffer_global_fadd_f64_inst, |
| 3329 | .fma_mix_insts, |
| 3330 | .fmacf64_inst, |
| 3331 | .full_rate_64_ops, |
| 3332 | .gfx9, |
| 3333 | .gfx90a_insts, |
| 3334 | .gfx940_insts, |
| 3335 | .kernarg_preload, |
| 3336 | .ldsbankcount32, |
| 3337 | .lshl_add_u64_inst, |
| 3338 | .mai_insts, |
| 3339 | .memory_atomic_fadd_f32_denormal_support, |
| 3340 | .packed_fp32_ops, |
| 3341 | .packed_tid, |
| 3342 | .pk_fmac_f16_inst, |
| 3343 | .requires_cov6, |
| 3344 | .sramecc_support, |
| 3345 | .vgpr_align2, |
| 3346 | }), |
| 3347 | }; |
| 3348 | pub const gfx9_generic: CpuModel = .{ |
| 3349 | .name = "gfx9_generic", |
| 3350 | .llvm_name = "gfx9-generic", |
| 3351 | .features = featureSet(&[_]Feature{ |
| 3352 | .addressablelocalmemorysize65536, |
| 3353 | .ds_src2_insts, |
| 3354 | .extended_image_insts, |
| 3355 | .gds, |
| 3356 | .gfx9, |
| 3357 | .image_gather4_d16_bug, |
| 3358 | .image_insts, |
| 3359 | .ldsbankcount32, |
| 3360 | .mad_mac_f32_insts, |
| 3361 | .requires_cov6, |
| 3362 | }), |
| 3363 | }; |
| 3364 | pub const hainan: CpuModel = .{ |
| 3365 | .name = "hainan", |
| 3366 | .llvm_name = "hainan", |
| 3367 | .features = featureSet(&[_]Feature{ |
| 3368 | .southern_islands, |
| 3369 | }), |
| 3370 | }; |
| 3371 | pub const hawaii: CpuModel = .{ |
| 3372 | .name = "hawaii", |
| 3373 | .llvm_name = "hawaii", |
| 3374 | .features = featureSet(&[_]Feature{ |
| 3375 | .fast_fmaf, |
| 3376 | .half_rate_64_ops, |
| 3377 | .ldsbankcount32, |
| 3378 | .sea_islands, |
| 3379 | }), |
| 3380 | }; |
| 3381 | pub const iceland: CpuModel = .{ |
| 3382 | .name = "iceland", |
| 3383 | .llvm_name = "iceland", |
| 3384 | .features = featureSet(&[_]Feature{ |
| 3385 | .ldsbankcount32, |
| 3386 | .sgpr_init_bug, |
| 3387 | .unpacked_d16_vmem, |
| 3388 | .volcanic_islands, |
| 3389 | }), |
| 3390 | }; |
| 3391 | pub const kabini: CpuModel = .{ |
| 3392 | .name = "kabini", |
| 3393 | .llvm_name = "kabini", |
| 3394 | .features = featureSet(&[_]Feature{ |
| 3395 | .ldsbankcount16, |
| 3396 | .sea_islands, |
| 3397 | }), |
| 3398 | }; |
| 3399 | pub const kaveri: CpuModel = .{ |
| 3400 | .name = "kaveri", |
| 3401 | .llvm_name = "kaveri", |
| 3402 | .features = featureSet(&[_]Feature{ |
| 3403 | .ldsbankcount32, |
| 3404 | .sea_islands, |
| 3405 | }), |
| 3406 | }; |
| 3407 | pub const mullins: CpuModel = .{ |
| 3408 | .name = "mullins", |
| 3409 | .llvm_name = "mullins", |
| 3410 | .features = featureSet(&[_]Feature{ |
| 3411 | .ldsbankcount16, |
| 3412 | .sea_islands, |
| 3413 | }), |
| 3414 | }; |
| 3415 | pub const oland: CpuModel = .{ |
| 3416 | .name = "oland", |
| 3417 | .llvm_name = "oland", |
| 3418 | .features = featureSet(&[_]Feature{ |
| 3419 | .southern_islands, |
| 3420 | }), |
| 3421 | }; |
| 3422 | pub const pitcairn: CpuModel = .{ |
| 3423 | .name = "pitcairn", |
| 3424 | .llvm_name = "pitcairn", |
| 3425 | .features = featureSet(&[_]Feature{ |
| 3426 | .southern_islands, |
| 3427 | }), |
| 3428 | }; |
| 3429 | pub const polaris10: CpuModel = .{ |
| 3430 | .name = "polaris10", |
| 3431 | .llvm_name = "polaris10", |
| 3432 | .features = featureSet(&[_]Feature{ |
| 3433 | .ldsbankcount32, |
| 3434 | .unpacked_d16_vmem, |
| 3435 | .volcanic_islands, |
| 3436 | }), |
| 3437 | }; |
| 3438 | pub const polaris11: CpuModel = .{ |
| 3439 | .name = "polaris11", |
| 3440 | .llvm_name = "polaris11", |
| 3441 | .features = featureSet(&[_]Feature{ |
| 3442 | .ldsbankcount32, |
| 3443 | .unpacked_d16_vmem, |
| 3444 | .volcanic_islands, |
| 3445 | }), |
| 3446 | }; |
| 3447 | pub const stoney: CpuModel = .{ |
| 3448 | .name = "stoney", |
| 3449 | .llvm_name = "stoney", |
| 3450 | .features = featureSet(&[_]Feature{ |
| 3451 | .image_gather4_d16_bug, |
| 3452 | .image_store_d16_bug, |
| 3453 | .ldsbankcount16, |
| 3454 | .volcanic_islands, |
| 3455 | .xnack_support, |
| 3456 | }), |
| 3457 | }; |
| 3458 | pub const tahiti: CpuModel = .{ |
| 3459 | .name = "tahiti", |
| 3460 | .llvm_name = "tahiti", |
| 3461 | .features = featureSet(&[_]Feature{ |
| 3462 | .fast_fmaf, |
| 3463 | .half_rate_64_ops, |
| 3464 | .southern_islands, |
| 3465 | }), |
| 3466 | }; |
| 3467 | pub const tonga: CpuModel = .{ |
| 3468 | .name = "tonga", |
| 3469 | .llvm_name = "tonga", |
| 3470 | .features = featureSet(&[_]Feature{ |
| 3471 | .ldsbankcount32, |
| 3472 | .sgpr_init_bug, |
| 3473 | .unpacked_d16_vmem, |
| 3474 | .volcanic_islands, |
| 3475 | }), |
| 3476 | }; |
| 3477 | pub const tongapro: CpuModel = .{ |
| 3478 | .name = "tongapro", |
| 3479 | .llvm_name = "tongapro", |
| 3480 | .features = featureSet(&[_]Feature{ |
| 3481 | .ldsbankcount32, |
| 3482 | .sgpr_init_bug, |
| 3483 | .unpacked_d16_vmem, |
| 3484 | .volcanic_islands, |
| 3485 | }), |
| 3486 | }; |
| 3487 | pub const verde: CpuModel = .{ |
| 3488 | .name = "verde", |
| 3489 | .llvm_name = "verde", |
| 3490 | .features = featureSet(&[_]Feature{ |
| 3491 | .southern_islands, |
| 3492 | }), |
| 3493 | }; |
| 3494 | }; |