-
Notifications
You must be signed in to change notification settings - Fork 13.4k
[NFC][AMDGPU] Auto generate check lines for some codegen tests #137534
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
Conversation
Make preparation for #137488.
@llvm/pr-subscribers-backend-amdgpu Author: Shilei Tian (shiltian) ChangesMake preparation for #137488. Patch is 314.42 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/137534.diff 11 Files Affected:
diff --git a/llvm/test/CodeGen/AMDGPU/dag-divergence.ll b/llvm/test/CodeGen/AMDGPU/dag-divergence.ll
index 9f83393d88061..cdf4a88814dfc 100644
--- a/llvm/test/CodeGen/AMDGPU/dag-divergence.ll
+++ b/llvm/test/CodeGen/AMDGPU/dag-divergence.ll
@@ -1,11 +1,29 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=fiji -verify-machineinstrs < %s | FileCheck -check-prefix=GCN %s
-; GCN-LABEL: {{^}}private_load_maybe_divergent:
-; GCN: buffer_load_dword
-; GCN-NOT: s_load_dword s
-; GCN: flat_load_dword
-; GCN-NOT: s_load_dword s
define amdgpu_kernel void @private_load_maybe_divergent(ptr addrspace(4) %k, ptr %flat) {
+; GCN-LABEL: private_load_maybe_divergent:
+; GCN: ; %bb.0:
+; GCN-NEXT: s_add_i32 s12, s12, s17
+; GCN-NEXT: s_mov_b64 s[22:23], s[2:3]
+; GCN-NEXT: s_lshr_b32 flat_scratch_hi, s12, 8
+; GCN-NEXT: s_mov_b64 s[20:21], s[0:1]
+; GCN-NEXT: s_add_u32 s20, s20, s17
+; GCN-NEXT: s_addc_u32 s21, s21, 0
+; GCN-NEXT: buffer_load_dword v0, v0, s[20:23], 0 offen glc
+; GCN-NEXT: s_waitcnt vmcnt(0)
+; GCN-NEXT: s_load_dwordx2 s[0:1], s[8:9], 0x0
+; GCN-NEXT: s_mov_b32 flat_scratch_lo, s13
+; GCN-NEXT: s_waitcnt lgkmcnt(0)
+; GCN-NEXT: v_mov_b32_e32 v2, s1
+; GCN-NEXT: v_ashrrev_i32_e32 v1, 31, v0
+; GCN-NEXT: v_lshlrev_b64 v[0:1], 2, v[0:1]
+; GCN-NEXT: v_add_u32_e32 v0, vcc, s0, v0
+; GCN-NEXT: v_addc_u32_e32 v1, vcc, v2, v1, vcc
+; GCN-NEXT: flat_load_dword v0, v[0:1]
+; GCN-NEXT: s_waitcnt vmcnt(0)
+; GCN-NEXT: flat_store_dword v[0:1], v0
+; GCN-NEXT: s_endpgm
%load = load volatile i32, ptr addrspace(5) poison, align 4
%gep = getelementptr inbounds i32, ptr addrspace(4) %k, i32 %load
%maybe.not.uniform.load = load i32, ptr addrspace(4) %gep, align 4
@@ -13,15 +31,27 @@ define amdgpu_kernel void @private_load_maybe_divergent(ptr addrspace(4) %k, ptr
ret void
}
-; GCN-LABEL: {{^}}flat_load_maybe_divergent:
-; GCN: s_load_dwordx4
-; GCN-NOT: s_load
-; GCN: flat_load_dword
-; GCN-NOT: s_load
-; GCN: flat_load_dword
-; GCN-NOT: s_load
-; GCN: flat_store_dword
define amdgpu_kernel void @flat_load_maybe_divergent(ptr addrspace(4) %k, ptr %flat) {
+; GCN-LABEL: flat_load_maybe_divergent:
+; GCN: ; %bb.0:
+; GCN-NEXT: s_load_dwordx4 s[0:3], s[8:9], 0x0
+; GCN-NEXT: s_add_i32 s12, s12, s17
+; GCN-NEXT: s_mov_b32 flat_scratch_lo, s13
+; GCN-NEXT: s_lshr_b32 flat_scratch_hi, s12, 8
+; GCN-NEXT: s_waitcnt lgkmcnt(0)
+; GCN-NEXT: v_mov_b32_e32 v0, s2
+; GCN-NEXT: v_mov_b32_e32 v1, s3
+; GCN-NEXT: flat_load_dword v0, v[0:1]
+; GCN-NEXT: v_mov_b32_e32 v2, s1
+; GCN-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GCN-NEXT: v_ashrrev_i32_e32 v1, 31, v0
+; GCN-NEXT: v_lshlrev_b64 v[0:1], 2, v[0:1]
+; GCN-NEXT: v_add_u32_e32 v0, vcc, s0, v0
+; GCN-NEXT: v_addc_u32_e32 v1, vcc, v2, v1, vcc
+; GCN-NEXT: flat_load_dword v0, v[0:1]
+; GCN-NEXT: s_waitcnt vmcnt(0)
+; GCN-NEXT: flat_store_dword v[0:1], v0
+; GCN-NEXT: s_endpgm
%load = load i32, ptr %flat, align 4
%gep = getelementptr inbounds i32, ptr addrspace(4) %k, i32 %load
%maybe.not.uniform.load = load i32, ptr addrspace(4) %gep, align 4
@@ -34,12 +64,33 @@ define amdgpu_kernel void @flat_load_maybe_divergent(ptr addrspace(4) %k, ptr %f
; last values are divergent due to the carry in glue (such that
; divergence needs to propagate through glue if there are any non-void
; outputs)
-; GCN-LABEL: {{^}}wide_carry_divergence_error:
-; GCN: v_sub_u32_e32
-; GCN: v_subb_u32_e32
-; GCN: v_subb_u32_e32
-; GCN: v_subb_u32_e32
define <2 x i128> @wide_carry_divergence_error(i128 %arg) {
+; GCN-LABEL: wide_carry_divergence_error:
+; GCN: ; %bb.0:
+; GCN-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GCN-NEXT: v_ffbh_u32_e32 v0, v0
+; GCN-NEXT: v_ffbh_u32_e32 v4, v2
+; GCN-NEXT: v_add_u32_e64 v0, s[4:5], v0, 32 clamp
+; GCN-NEXT: v_ffbh_u32_e32 v1, v1
+; GCN-NEXT: v_add_u32_e32 v4, vcc, 32, v4
+; GCN-NEXT: v_min3_u32 v0, v0, v1, 64
+; GCN-NEXT: v_add_u32_e32 v0, vcc, 64, v0
+; GCN-NEXT: v_ffbh_u32_e32 v5, v3
+; GCN-NEXT: v_addc_u32_e64 v1, s[4:5], 0, 0, vcc
+; GCN-NEXT: v_cmp_ne_u64_e32 vcc, 0, v[2:3]
+; GCN-NEXT: v_min_u32_e32 v4, v4, v5
+; GCN-NEXT: v_cndmask_b32_e32 v0, v0, v4, vcc
+; GCN-NEXT: v_cndmask_b32_e64 v1, v1, 0, vcc
+; GCN-NEXT: v_sub_u32_e32 v0, vcc, 0, v0
+; GCN-NEXT: v_mov_b32_e32 v3, 0
+; GCN-NEXT: v_subb_u32_e32 v1, vcc, 0, v1, vcc
+; GCN-NEXT: v_subb_u32_e32 v2, vcc, 0, v3, vcc
+; GCN-NEXT: v_subb_u32_e32 v3, vcc, 0, v3, vcc
+; GCN-NEXT: v_mov_b32_e32 v4, 0
+; GCN-NEXT: v_mov_b32_e32 v5, 0
+; GCN-NEXT: v_mov_b32_e32 v6, 0
+; GCN-NEXT: v_mov_b32_e32 v7, 0
+; GCN-NEXT: s_setpc_b64 s[30:31]
%i = call i128 @llvm.ctlz.i128(i128 %arg, i1 false)
%i1 = sub i128 0, %i
%i2 = insertelement <2 x i128> zeroinitializer, i128 %i1, i64 0
diff --git a/llvm/test/CodeGen/AMDGPU/flat-offset-bug.ll b/llvm/test/CodeGen/AMDGPU/flat-offset-bug.ll
index 54343fa820cba..1732dd0521e5f 100644
--- a/llvm/test/CodeGen/AMDGPU/flat-offset-bug.ll
+++ b/llvm/test/CodeGen/AMDGPU/flat-offset-bug.ll
@@ -1,13 +1,40 @@
-; RUN: llc -mtriple=amdgcn -mcpu=gfx900 -verify-machineinstrs < %s | FileCheck -check-prefixes=GCN,GFX9_11 %s
-; RUN: llc -mtriple=amdgcn -mcpu=gfx1010 -verify-machineinstrs < %s | FileCheck -check-prefixes=GCN,GFX10 %s
-; RUN: llc -mtriple=amdgcn -mcpu=gfx1100 -verify-machineinstrs < %s | FileCheck -check-prefixes=GCN,GFX9_11 %s
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
+; RUN: llc -mtriple=amdgcn -mcpu=gfx900 -verify-machineinstrs < %s | FileCheck -check-prefix=GFX9 %s
+; RUN: llc -mtriple=amdgcn -mcpu=gfx1010 -verify-machineinstrs < %s | FileCheck -check-prefix=GFX10 %s
+; RUN: llc -mtriple=amdgcn -mcpu=gfx1100 -verify-machineinstrs < %s | FileCheck -check-prefix=GFX11 %s
-; GCN-LABEL: flat_inst_offset:
-; GFX9_11: flat_load_{{dword|b32}} v{{[0-9]+}}, v[{{[0-9:]+}}] offset:4
-; GFX9_11: flat_store_{{dword|b32}} v[{{[0-9:]+}}], v{{[0-9]+}} offset:4
-; GFX10: flat_load_dword v{{[0-9]+}}, v[{{[0-9:]+}}]{{$}}
-; GFX10: flat_store_dword v[{{[0-9:]+}}], v{{[0-9]+}}{{$}}
define void @flat_inst_offset(ptr nocapture %p) {
+; GFX9-LABEL: flat_inst_offset:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX9-NEXT: flat_load_dword v2, v[0:1] offset:4
+; GFX9-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX9-NEXT: v_add_u32_e32 v2, 1, v2
+; GFX9-NEXT: flat_store_dword v[0:1], v2 offset:4
+; GFX9-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX9-NEXT: s_setpc_b64 s[30:31]
+;
+; GFX10-LABEL: flat_inst_offset:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX10-NEXT: v_add_co_u32 v0, vcc_lo, v0, 4
+; GFX10-NEXT: v_add_co_ci_u32_e32 v1, vcc_lo, 0, v1, vcc_lo
+; GFX10-NEXT: flat_load_dword v2, v[0:1]
+; GFX10-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX10-NEXT: v_add_nc_u32_e32 v2, 1, v2
+; GFX10-NEXT: flat_store_dword v[0:1], v2
+; GFX10-NEXT: s_waitcnt lgkmcnt(0)
+; GFX10-NEXT: s_setpc_b64 s[30:31]
+;
+; GFX11-LABEL: flat_inst_offset:
+; GFX11: ; %bb.0:
+; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX11-NEXT: flat_load_b32 v2, v[0:1] offset:4
+; GFX11-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX11-NEXT: v_add_nc_u32_e32 v2, 1, v2
+; GFX11-NEXT: flat_store_b32 v[0:1], v2 offset:4
+; GFX11-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-NEXT: s_setpc_b64 s[30:31]
%gep = getelementptr inbounds i32, ptr %p, i64 1
%load = load i32, ptr %gep, align 4
%inc = add nsw i32 %load, 1
@@ -15,10 +42,34 @@ define void @flat_inst_offset(ptr nocapture %p) {
ret void
}
-; GCN-LABEL: global_inst_offset:
-; GCN: global_load_{{dword|b32}} v{{[0-9]+}}, v[{{[0-9:]+}}], off offset:4
-; GCN: global_store_{{dword|b32}} v[{{[0-9:]+}}], v{{[0-9]+}}, off offset:4
define void @global_inst_offset(ptr addrspace(1) nocapture %p) {
+; GFX9-LABEL: global_inst_offset:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX9-NEXT: global_load_dword v2, v[0:1], off offset:4
+; GFX9-NEXT: s_waitcnt vmcnt(0)
+; GFX9-NEXT: v_add_u32_e32 v2, 1, v2
+; GFX9-NEXT: global_store_dword v[0:1], v2, off offset:4
+; GFX9-NEXT: s_waitcnt vmcnt(0)
+; GFX9-NEXT: s_setpc_b64 s[30:31]
+;
+; GFX10-LABEL: global_inst_offset:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX10-NEXT: global_load_dword v2, v[0:1], off offset:4
+; GFX10-NEXT: s_waitcnt vmcnt(0)
+; GFX10-NEXT: v_add_nc_u32_e32 v2, 1, v2
+; GFX10-NEXT: global_store_dword v[0:1], v2, off offset:4
+; GFX10-NEXT: s_setpc_b64 s[30:31]
+;
+; GFX11-LABEL: global_inst_offset:
+; GFX11: ; %bb.0:
+; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX11-NEXT: global_load_b32 v2, v[0:1], off offset:4
+; GFX11-NEXT: s_waitcnt vmcnt(0)
+; GFX11-NEXT: v_add_nc_u32_e32 v2, 1, v2
+; GFX11-NEXT: global_store_b32 v[0:1], v2, off offset:4
+; GFX11-NEXT: s_setpc_b64 s[30:31]
%gep = getelementptr inbounds i32, ptr addrspace(1) %p, i64 1
%load = load i32, ptr addrspace(1) %gep, align 4
%inc = add nsw i32 %load, 1
@@ -26,10 +77,51 @@ define void @global_inst_offset(ptr addrspace(1) nocapture %p) {
ret void
}
-; GCN-LABEL: load_i16_lo:
-; GFX9_11: flat_load_{{short_d16|d16_b16}} v{{[0-9]+}}, v[{{[0-9:]+}}] offset:8{{$}}
-; GFX10: flat_load_short_d16 v{{[0-9]+}}, v[{{[0-9:]+}}]{{$}}
define amdgpu_kernel void @load_i16_lo(ptr %arg, ptr %out) {
+; GFX9-LABEL: load_i16_lo:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
+; GFX9-NEXT: v_mov_b32_e32 v2, 0
+; GFX9-NEXT: s_waitcnt lgkmcnt(0)
+; GFX9-NEXT: v_mov_b32_e32 v0, s0
+; GFX9-NEXT: v_mov_b32_e32 v1, s1
+; GFX9-NEXT: flat_load_short_d16 v2, v[0:1] offset:8
+; GFX9-NEXT: v_mov_b32_e32 v0, s2
+; GFX9-NEXT: v_mov_b32_e32 v1, s3
+; GFX9-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX9-NEXT: v_pk_add_u16 v2, v2, v2
+; GFX9-NEXT: flat_store_dword v[0:1], v2
+; GFX9-NEXT: s_endpgm
+;
+; GFX10-LABEL: load_i16_lo:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
+; GFX10-NEXT: v_mov_b32_e32 v2, 0
+; GFX10-NEXT: s_waitcnt lgkmcnt(0)
+; GFX10-NEXT: s_add_u32 s0, s0, 8
+; GFX10-NEXT: s_addc_u32 s1, s1, 0
+; GFX10-NEXT: v_mov_b32_e32 v0, s0
+; GFX10-NEXT: v_mov_b32_e32 v1, s1
+; GFX10-NEXT: flat_load_short_d16 v2, v[0:1]
+; GFX10-NEXT: v_mov_b32_e32 v0, s2
+; GFX10-NEXT: v_mov_b32_e32 v1, s3
+; GFX10-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX10-NEXT: v_pk_add_u16 v2, v2, v2
+; GFX10-NEXT: flat_store_dword v[0:1], v2
+; GFX10-NEXT: s_endpgm
+;
+; GFX11-LABEL: load_i16_lo:
+; GFX11: ; %bb.0:
+; GFX11-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
+; GFX11-NEXT: v_mov_b32_e32 v2, 0
+; GFX11-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
+; GFX11-NEXT: flat_load_d16_b16 v2, v[0:1] offset:8
+; GFX11-NEXT: v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, s3
+; GFX11-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX11-NEXT: v_pk_add_u16 v2, v2, v2
+; GFX11-NEXT: flat_store_b32 v[0:1], v2
+; GFX11-NEXT: s_endpgm
%gep = getelementptr inbounds i16, ptr %arg, i32 4
%ld = load i16, ptr %gep, align 2
%vec = insertelement <2 x i16> <i16 poison, i16 0>, i16 %ld, i32 0
@@ -38,10 +130,51 @@ define amdgpu_kernel void @load_i16_lo(ptr %arg, ptr %out) {
ret void
}
-; GCN-LABEL: load_i16_hi:
-; GFX9_11: flat_load_{{short_d16_hi|d16_hi_b16}} v{{[0-9]+}}, v[{{[0-9:]+}}] offset:8{{$}}
-; GFX10: flat_load_short_d16_hi v{{[0-9]+}}, v[{{[0-9:]+}}]{{$}}
define amdgpu_kernel void @load_i16_hi(ptr %arg, ptr %out) {
+; GFX9-LABEL: load_i16_hi:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
+; GFX9-NEXT: v_mov_b32_e32 v2, 0
+; GFX9-NEXT: s_waitcnt lgkmcnt(0)
+; GFX9-NEXT: v_mov_b32_e32 v0, s0
+; GFX9-NEXT: v_mov_b32_e32 v1, s1
+; GFX9-NEXT: flat_load_short_d16_hi v2, v[0:1] offset:8
+; GFX9-NEXT: v_mov_b32_e32 v0, s2
+; GFX9-NEXT: v_mov_b32_e32 v1, s3
+; GFX9-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX9-NEXT: v_pk_add_u16 v2, v2, v2
+; GFX9-NEXT: flat_store_dword v[0:1], v2
+; GFX9-NEXT: s_endpgm
+;
+; GFX10-LABEL: load_i16_hi:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
+; GFX10-NEXT: v_mov_b32_e32 v2, 0
+; GFX10-NEXT: s_waitcnt lgkmcnt(0)
+; GFX10-NEXT: s_add_u32 s0, s0, 8
+; GFX10-NEXT: s_addc_u32 s1, s1, 0
+; GFX10-NEXT: v_mov_b32_e32 v0, s0
+; GFX10-NEXT: v_mov_b32_e32 v1, s1
+; GFX10-NEXT: flat_load_short_d16_hi v2, v[0:1]
+; GFX10-NEXT: v_mov_b32_e32 v0, s2
+; GFX10-NEXT: v_mov_b32_e32 v1, s3
+; GFX10-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX10-NEXT: v_pk_add_u16 v2, v2, v2
+; GFX10-NEXT: flat_store_dword v[0:1], v2
+; GFX10-NEXT: s_endpgm
+;
+; GFX11-LABEL: load_i16_hi:
+; GFX11: ; %bb.0:
+; GFX11-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
+; GFX11-NEXT: v_mov_b32_e32 v2, 0
+; GFX11-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
+; GFX11-NEXT: flat_load_d16_hi_b16 v2, v[0:1] offset:8
+; GFX11-NEXT: v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, s3
+; GFX11-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX11-NEXT: v_pk_add_u16 v2, v2, v2
+; GFX11-NEXT: flat_store_b32 v[0:1], v2
+; GFX11-NEXT: s_endpgm
%gep = getelementptr inbounds i16, ptr %arg, i32 4
%ld = load i16, ptr %gep, align 2
%vec = insertelement <2 x i16> <i16 0, i16 poison>, i16 %ld, i32 1
@@ -50,10 +183,51 @@ define amdgpu_kernel void @load_i16_hi(ptr %arg, ptr %out) {
ret void
}
-; GCN-LABEL: load_half_lo:
-; GFX9_11: flat_load_{{short_d16|d16_b16}} v{{[0-9]+}}, v[{{[0-9:]+}}] offset:8{{$}}
-; GFX10: flat_load_short_d16 v{{[0-9]+}}, v[{{[0-9:]+}}]{{$}}
define amdgpu_kernel void @load_half_lo(ptr %arg, ptr %out) {
+; GFX9-LABEL: load_half_lo:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
+; GFX9-NEXT: v_mov_b32_e32 v2, 0
+; GFX9-NEXT: s_waitcnt lgkmcnt(0)
+; GFX9-NEXT: v_mov_b32_e32 v0, s0
+; GFX9-NEXT: v_mov_b32_e32 v1, s1
+; GFX9-NEXT: flat_load_short_d16 v2, v[0:1] offset:8
+; GFX9-NEXT: v_mov_b32_e32 v0, s2
+; GFX9-NEXT: v_mov_b32_e32 v1, s3
+; GFX9-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX9-NEXT: v_pk_add_f16 v2, v2, v2
+; GFX9-NEXT: flat_store_dword v[0:1], v2
+; GFX9-NEXT: s_endpgm
+;
+; GFX10-LABEL: load_half_lo:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
+; GFX10-NEXT: v_mov_b32_e32 v2, 0
+; GFX10-NEXT: s_waitcnt lgkmcnt(0)
+; GFX10-NEXT: s_add_u32 s0, s0, 8
+; GFX10-NEXT: s_addc_u32 s1, s1, 0
+; GFX10-NEXT: v_mov_b32_e32 v0, s0
+; GFX10-NEXT: v_mov_b32_e32 v1, s1
+; GFX10-NEXT: flat_load_short_d16 v2, v[0:1]
+; GFX10-NEXT: v_mov_b32_e32 v0, s2
+; GFX10-NEXT: v_mov_b32_e32 v1, s3
+; GFX10-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX10-NEXT: v_pk_add_f16 v2, v2, v2
+; GFX10-NEXT: flat_store_dword v[0:1], v2
+; GFX10-NEXT: s_endpgm
+;
+; GFX11-LABEL: load_half_lo:
+; GFX11: ; %bb.0:
+; GFX11-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
+; GFX11-NEXT: v_mov_b32_e32 v2, 0
+; GFX11-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
+; GFX11-NEXT: flat_load_d16_b16 v2, v[0:1] offset:8
+; GFX11-NEXT: v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, s3
+; GFX11-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX11-NEXT: v_pk_add_f16 v2, v2, v2
+; GFX11-NEXT: flat_store_b32 v[0:1], v2
+; GFX11-NEXT: s_endpgm
%gep = getelementptr inbounds half, ptr %arg, i32 4
%ld = load half, ptr %gep, align 2
%vec = insertelement <2 x half> <half poison, half 0xH0000>, half %ld, i32 0
@@ -62,10 +236,51 @@ define amdgpu_kernel void @load_half_lo(ptr %arg, ptr %out) {
ret void
}
-; GCN-LABEL: load_half_hi:
-; GFX9_11: flat_load_{{short_d16_hi|d16_hi_b16}} v{{[0-9]+}}, v[{{[0-9:]+}}] offset:8{{$}}
-; GFX10: flat_load_short_d16_hi v{{[0-9]+}}, v[{{[0-9:]+}}]{{$}}
define amdgpu_kernel void @load_half_hi(ptr %arg, ptr %out) {
+; GFX9-LABEL: load_half_hi:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
+; GFX9-NEXT: v_mov_b32_e32 v2, 0
+; GFX9-NEXT: s_waitcnt lgkmcnt(0)
+; GFX9-NEXT: v_mov_b32_e32 v0, s0
+; GFX9-NEXT: v_mov_b32_e32 v1, s1
+; GFX9-NEXT: flat_load_short_d16_hi v2, v[0:1] offset:8
+; GFX9-NEXT: v_mov_b32_e32 v0, s2
+; GFX9-NEXT: v_mov_b32_e32 v1, s3
+; GFX9-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX9-NEXT: v_pk_add_f16 v2, v2, v2
+; GFX9-NEXT: flat_store_dword v[0:1], v2
+; GFX9-NEXT: s_endpgm
+;
+; GFX10-LABEL: load_half_hi:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
+; GFX10-NEXT: v_mov_b32_e32 v2, 0
+; GFX10-NEXT: s_waitcnt lgkmcnt(0)
+; GFX10-NEXT: s_add_u32 s0, s0, 8
+; GFX10-NEXT: s_addc_u32 s1, s1, 0
+; GFX10-NEXT: v_mov_b32_e32 v0, s0
+; GFX10-NEXT: v_mov_b32_e32 v1, s1
+; GFX10-NEXT: flat_load_short_d16_hi v2, v[0:1]
+; GFX10-NEXT: v_mov_b32_e32 v0, s2
+; GFX10-NEXT: v_mov_b32_e32 v1, s3
+; GFX10-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX10-NEXT: v_pk_add_f16 v2, v2, v2
+; GFX10-NEXT: flat_store_dword v[0:1], v2
+; GFX10-NEXT: s_endpgm
+;
+; GFX11-LABEL: load_half_hi:
+; GFX11: ; %bb.0:
+; GFX11-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
+; GFX11-NEXT: v_mov_b32_e32 v2, 0
+; GFX11-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
+; GFX11-NEXT: flat_load_d16_hi_b16 v2, v[0:1] offset:8
+; GFX11-NEXT: v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, s3
+; GFX11-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX11-NEXT: v_pk_add_f16 v2, v2, v2
+; GFX11-NEXT: flat_store_b32 v[0:1], v2
+; GFX11-NEXT: s_endpgm
%gep = getelementptr inbounds half, ptr %arg, i32 4
%ld = load half, ptr %gep, align 2
%vec = insertelement <2 x half> <half 0xH0000, half poison>, half %ld, i32 1
@@ -74,10 +289,48 @@ define amdgpu_kernel void @load_half_hi(ptr %arg, ptr %out) {
ret void
}
-; GCN-LABEL: load_float_lo:
-; GFX9_11: flat_load_{{dword|b32}} v{{[0-9]+}}, v[{{[0-9:]+}}] offset:16{{$}}
-; GFX10: flat_load_dword v{{[0-9]+}}, v[{{[0-9:]+}}]{{$}}
define amdgpu_kernel void @load_float_lo(ptr %arg, ptr %out) {
+; GFX9-LABEL: load_float_lo:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
+; GFX9-NEXT: s_waitcnt lgkmcnt(0)
+; GFX9-NEXT: v_mov_b32_e32 v0, s0
+; GFX9-NEXT: v_mov_b32_e32 v1, s1
+; GFX9-NEXT: flat_load_dword v2, v[0:1] offset:16
+; GFX9-NEXT: v_mov_b32_e32 v0, s2
+; GFX9-NEXT: v_mov_b32_e32 v1, s3
+; GFX9-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX9-NEXT: v_add_f32_e32 v2, v2, v2
+; GFX9-NEXT: flat_store_dword v[0:1], v2
+; GFX9-NEXT: s_endpgm
+;
+; GFX10-LABEL: load_float_lo:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
+; GFX10-NEXT: s_waitcnt lgkmcnt(0)
+; GFX10-NEXT: s_add_u32 s0, s0, 16
+; GFX10-NEXT: s_addc_u32 s1, s1, 0
+; GFX10-NEXT: v_mov_b32_e32 v0, s0
+; GFX10-NEXT: v_mov_b32_e32 v1, s1
+; GFX10-NEXT: flat_load_dword v2, v[0:1]
+; GFX10-NEXT: v_mov_b32_e32 v0, s2
+; GFX10-NEXT: v_mov_b32_e32 v1, s3
+; GFX10-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX10-NEXT: v_add_f32_e32 v2, v2, v2
+; GFX10-NEXT: flat_store_dword v[0:1], v2
+; GFX10-NEXT: s_endpgm
+;
+; GFX11-LABEL: load_float_lo:
+; GFX11: ; %bb.0:
+; GFX11-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
+; GFX11-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
+; GFX11-NEXT: flat_load_b32 v2, v[0:1] offset:16
+; GFX11-NEXT: v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, s3
+; GFX11-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX11-NEXT: v_add_f32_e32 v2, v2, v2
+; GFX11-NEXT: flat_store_b32 v[0:1], v2
+; GFX11-NEXT: s_endpgm
%gep = getelementptr inbounds float, ptr %arg, i32 4
%ld = load float, ptr %gep, align 4
%v = fadd float %ld, %ld
diff --git a/llvm/test/CodeGen/AMDGPU/lds-misaligned-bug.ll b/llvm/test/CodeGen/AMDGPU/lds-misaligned-bug.ll
index 278ad63b0b76c..7ffc2a6987742 100644
--- a/llvm/...
[truncated]
|
@llvm/pr-subscribers-llvm-transforms Author: Shilei Tian (shiltian) ChangesMake preparation for #137488. Patch is 314.42 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/137534.diff 11 Files Affected:
diff --git a/llvm/test/CodeGen/AMDGPU/dag-divergence.ll b/llvm/test/CodeGen/AMDGPU/dag-divergence.ll
index 9f83393d88061..cdf4a88814dfc 100644
--- a/llvm/test/CodeGen/AMDGPU/dag-divergence.ll
+++ b/llvm/test/CodeGen/AMDGPU/dag-divergence.ll
@@ -1,11 +1,29 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=fiji -verify-machineinstrs < %s | FileCheck -check-prefix=GCN %s
-; GCN-LABEL: {{^}}private_load_maybe_divergent:
-; GCN: buffer_load_dword
-; GCN-NOT: s_load_dword s
-; GCN: flat_load_dword
-; GCN-NOT: s_load_dword s
define amdgpu_kernel void @private_load_maybe_divergent(ptr addrspace(4) %k, ptr %flat) {
+; GCN-LABEL: private_load_maybe_divergent:
+; GCN: ; %bb.0:
+; GCN-NEXT: s_add_i32 s12, s12, s17
+; GCN-NEXT: s_mov_b64 s[22:23], s[2:3]
+; GCN-NEXT: s_lshr_b32 flat_scratch_hi, s12, 8
+; GCN-NEXT: s_mov_b64 s[20:21], s[0:1]
+; GCN-NEXT: s_add_u32 s20, s20, s17
+; GCN-NEXT: s_addc_u32 s21, s21, 0
+; GCN-NEXT: buffer_load_dword v0, v0, s[20:23], 0 offen glc
+; GCN-NEXT: s_waitcnt vmcnt(0)
+; GCN-NEXT: s_load_dwordx2 s[0:1], s[8:9], 0x0
+; GCN-NEXT: s_mov_b32 flat_scratch_lo, s13
+; GCN-NEXT: s_waitcnt lgkmcnt(0)
+; GCN-NEXT: v_mov_b32_e32 v2, s1
+; GCN-NEXT: v_ashrrev_i32_e32 v1, 31, v0
+; GCN-NEXT: v_lshlrev_b64 v[0:1], 2, v[0:1]
+; GCN-NEXT: v_add_u32_e32 v0, vcc, s0, v0
+; GCN-NEXT: v_addc_u32_e32 v1, vcc, v2, v1, vcc
+; GCN-NEXT: flat_load_dword v0, v[0:1]
+; GCN-NEXT: s_waitcnt vmcnt(0)
+; GCN-NEXT: flat_store_dword v[0:1], v0
+; GCN-NEXT: s_endpgm
%load = load volatile i32, ptr addrspace(5) poison, align 4
%gep = getelementptr inbounds i32, ptr addrspace(4) %k, i32 %load
%maybe.not.uniform.load = load i32, ptr addrspace(4) %gep, align 4
@@ -13,15 +31,27 @@ define amdgpu_kernel void @private_load_maybe_divergent(ptr addrspace(4) %k, ptr
ret void
}
-; GCN-LABEL: {{^}}flat_load_maybe_divergent:
-; GCN: s_load_dwordx4
-; GCN-NOT: s_load
-; GCN: flat_load_dword
-; GCN-NOT: s_load
-; GCN: flat_load_dword
-; GCN-NOT: s_load
-; GCN: flat_store_dword
define amdgpu_kernel void @flat_load_maybe_divergent(ptr addrspace(4) %k, ptr %flat) {
+; GCN-LABEL: flat_load_maybe_divergent:
+; GCN: ; %bb.0:
+; GCN-NEXT: s_load_dwordx4 s[0:3], s[8:9], 0x0
+; GCN-NEXT: s_add_i32 s12, s12, s17
+; GCN-NEXT: s_mov_b32 flat_scratch_lo, s13
+; GCN-NEXT: s_lshr_b32 flat_scratch_hi, s12, 8
+; GCN-NEXT: s_waitcnt lgkmcnt(0)
+; GCN-NEXT: v_mov_b32_e32 v0, s2
+; GCN-NEXT: v_mov_b32_e32 v1, s3
+; GCN-NEXT: flat_load_dword v0, v[0:1]
+; GCN-NEXT: v_mov_b32_e32 v2, s1
+; GCN-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GCN-NEXT: v_ashrrev_i32_e32 v1, 31, v0
+; GCN-NEXT: v_lshlrev_b64 v[0:1], 2, v[0:1]
+; GCN-NEXT: v_add_u32_e32 v0, vcc, s0, v0
+; GCN-NEXT: v_addc_u32_e32 v1, vcc, v2, v1, vcc
+; GCN-NEXT: flat_load_dword v0, v[0:1]
+; GCN-NEXT: s_waitcnt vmcnt(0)
+; GCN-NEXT: flat_store_dword v[0:1], v0
+; GCN-NEXT: s_endpgm
%load = load i32, ptr %flat, align 4
%gep = getelementptr inbounds i32, ptr addrspace(4) %k, i32 %load
%maybe.not.uniform.load = load i32, ptr addrspace(4) %gep, align 4
@@ -34,12 +64,33 @@ define amdgpu_kernel void @flat_load_maybe_divergent(ptr addrspace(4) %k, ptr %f
; last values are divergent due to the carry in glue (such that
; divergence needs to propagate through glue if there are any non-void
; outputs)
-; GCN-LABEL: {{^}}wide_carry_divergence_error:
-; GCN: v_sub_u32_e32
-; GCN: v_subb_u32_e32
-; GCN: v_subb_u32_e32
-; GCN: v_subb_u32_e32
define <2 x i128> @wide_carry_divergence_error(i128 %arg) {
+; GCN-LABEL: wide_carry_divergence_error:
+; GCN: ; %bb.0:
+; GCN-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GCN-NEXT: v_ffbh_u32_e32 v0, v0
+; GCN-NEXT: v_ffbh_u32_e32 v4, v2
+; GCN-NEXT: v_add_u32_e64 v0, s[4:5], v0, 32 clamp
+; GCN-NEXT: v_ffbh_u32_e32 v1, v1
+; GCN-NEXT: v_add_u32_e32 v4, vcc, 32, v4
+; GCN-NEXT: v_min3_u32 v0, v0, v1, 64
+; GCN-NEXT: v_add_u32_e32 v0, vcc, 64, v0
+; GCN-NEXT: v_ffbh_u32_e32 v5, v3
+; GCN-NEXT: v_addc_u32_e64 v1, s[4:5], 0, 0, vcc
+; GCN-NEXT: v_cmp_ne_u64_e32 vcc, 0, v[2:3]
+; GCN-NEXT: v_min_u32_e32 v4, v4, v5
+; GCN-NEXT: v_cndmask_b32_e32 v0, v0, v4, vcc
+; GCN-NEXT: v_cndmask_b32_e64 v1, v1, 0, vcc
+; GCN-NEXT: v_sub_u32_e32 v0, vcc, 0, v0
+; GCN-NEXT: v_mov_b32_e32 v3, 0
+; GCN-NEXT: v_subb_u32_e32 v1, vcc, 0, v1, vcc
+; GCN-NEXT: v_subb_u32_e32 v2, vcc, 0, v3, vcc
+; GCN-NEXT: v_subb_u32_e32 v3, vcc, 0, v3, vcc
+; GCN-NEXT: v_mov_b32_e32 v4, 0
+; GCN-NEXT: v_mov_b32_e32 v5, 0
+; GCN-NEXT: v_mov_b32_e32 v6, 0
+; GCN-NEXT: v_mov_b32_e32 v7, 0
+; GCN-NEXT: s_setpc_b64 s[30:31]
%i = call i128 @llvm.ctlz.i128(i128 %arg, i1 false)
%i1 = sub i128 0, %i
%i2 = insertelement <2 x i128> zeroinitializer, i128 %i1, i64 0
diff --git a/llvm/test/CodeGen/AMDGPU/flat-offset-bug.ll b/llvm/test/CodeGen/AMDGPU/flat-offset-bug.ll
index 54343fa820cba..1732dd0521e5f 100644
--- a/llvm/test/CodeGen/AMDGPU/flat-offset-bug.ll
+++ b/llvm/test/CodeGen/AMDGPU/flat-offset-bug.ll
@@ -1,13 +1,40 @@
-; RUN: llc -mtriple=amdgcn -mcpu=gfx900 -verify-machineinstrs < %s | FileCheck -check-prefixes=GCN,GFX9_11 %s
-; RUN: llc -mtriple=amdgcn -mcpu=gfx1010 -verify-machineinstrs < %s | FileCheck -check-prefixes=GCN,GFX10 %s
-; RUN: llc -mtriple=amdgcn -mcpu=gfx1100 -verify-machineinstrs < %s | FileCheck -check-prefixes=GCN,GFX9_11 %s
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
+; RUN: llc -mtriple=amdgcn -mcpu=gfx900 -verify-machineinstrs < %s | FileCheck -check-prefix=GFX9 %s
+; RUN: llc -mtriple=amdgcn -mcpu=gfx1010 -verify-machineinstrs < %s | FileCheck -check-prefix=GFX10 %s
+; RUN: llc -mtriple=amdgcn -mcpu=gfx1100 -verify-machineinstrs < %s | FileCheck -check-prefix=GFX11 %s
-; GCN-LABEL: flat_inst_offset:
-; GFX9_11: flat_load_{{dword|b32}} v{{[0-9]+}}, v[{{[0-9:]+}}] offset:4
-; GFX9_11: flat_store_{{dword|b32}} v[{{[0-9:]+}}], v{{[0-9]+}} offset:4
-; GFX10: flat_load_dword v{{[0-9]+}}, v[{{[0-9:]+}}]{{$}}
-; GFX10: flat_store_dword v[{{[0-9:]+}}], v{{[0-9]+}}{{$}}
define void @flat_inst_offset(ptr nocapture %p) {
+; GFX9-LABEL: flat_inst_offset:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX9-NEXT: flat_load_dword v2, v[0:1] offset:4
+; GFX9-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX9-NEXT: v_add_u32_e32 v2, 1, v2
+; GFX9-NEXT: flat_store_dword v[0:1], v2 offset:4
+; GFX9-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX9-NEXT: s_setpc_b64 s[30:31]
+;
+; GFX10-LABEL: flat_inst_offset:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX10-NEXT: v_add_co_u32 v0, vcc_lo, v0, 4
+; GFX10-NEXT: v_add_co_ci_u32_e32 v1, vcc_lo, 0, v1, vcc_lo
+; GFX10-NEXT: flat_load_dword v2, v[0:1]
+; GFX10-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX10-NEXT: v_add_nc_u32_e32 v2, 1, v2
+; GFX10-NEXT: flat_store_dword v[0:1], v2
+; GFX10-NEXT: s_waitcnt lgkmcnt(0)
+; GFX10-NEXT: s_setpc_b64 s[30:31]
+;
+; GFX11-LABEL: flat_inst_offset:
+; GFX11: ; %bb.0:
+; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX11-NEXT: flat_load_b32 v2, v[0:1] offset:4
+; GFX11-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX11-NEXT: v_add_nc_u32_e32 v2, 1, v2
+; GFX11-NEXT: flat_store_b32 v[0:1], v2 offset:4
+; GFX11-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-NEXT: s_setpc_b64 s[30:31]
%gep = getelementptr inbounds i32, ptr %p, i64 1
%load = load i32, ptr %gep, align 4
%inc = add nsw i32 %load, 1
@@ -15,10 +42,34 @@ define void @flat_inst_offset(ptr nocapture %p) {
ret void
}
-; GCN-LABEL: global_inst_offset:
-; GCN: global_load_{{dword|b32}} v{{[0-9]+}}, v[{{[0-9:]+}}], off offset:4
-; GCN: global_store_{{dword|b32}} v[{{[0-9:]+}}], v{{[0-9]+}}, off offset:4
define void @global_inst_offset(ptr addrspace(1) nocapture %p) {
+; GFX9-LABEL: global_inst_offset:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX9-NEXT: global_load_dword v2, v[0:1], off offset:4
+; GFX9-NEXT: s_waitcnt vmcnt(0)
+; GFX9-NEXT: v_add_u32_e32 v2, 1, v2
+; GFX9-NEXT: global_store_dword v[0:1], v2, off offset:4
+; GFX9-NEXT: s_waitcnt vmcnt(0)
+; GFX9-NEXT: s_setpc_b64 s[30:31]
+;
+; GFX10-LABEL: global_inst_offset:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX10-NEXT: global_load_dword v2, v[0:1], off offset:4
+; GFX10-NEXT: s_waitcnt vmcnt(0)
+; GFX10-NEXT: v_add_nc_u32_e32 v2, 1, v2
+; GFX10-NEXT: global_store_dword v[0:1], v2, off offset:4
+; GFX10-NEXT: s_setpc_b64 s[30:31]
+;
+; GFX11-LABEL: global_inst_offset:
+; GFX11: ; %bb.0:
+; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX11-NEXT: global_load_b32 v2, v[0:1], off offset:4
+; GFX11-NEXT: s_waitcnt vmcnt(0)
+; GFX11-NEXT: v_add_nc_u32_e32 v2, 1, v2
+; GFX11-NEXT: global_store_b32 v[0:1], v2, off offset:4
+; GFX11-NEXT: s_setpc_b64 s[30:31]
%gep = getelementptr inbounds i32, ptr addrspace(1) %p, i64 1
%load = load i32, ptr addrspace(1) %gep, align 4
%inc = add nsw i32 %load, 1
@@ -26,10 +77,51 @@ define void @global_inst_offset(ptr addrspace(1) nocapture %p) {
ret void
}
-; GCN-LABEL: load_i16_lo:
-; GFX9_11: flat_load_{{short_d16|d16_b16}} v{{[0-9]+}}, v[{{[0-9:]+}}] offset:8{{$}}
-; GFX10: flat_load_short_d16 v{{[0-9]+}}, v[{{[0-9:]+}}]{{$}}
define amdgpu_kernel void @load_i16_lo(ptr %arg, ptr %out) {
+; GFX9-LABEL: load_i16_lo:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
+; GFX9-NEXT: v_mov_b32_e32 v2, 0
+; GFX9-NEXT: s_waitcnt lgkmcnt(0)
+; GFX9-NEXT: v_mov_b32_e32 v0, s0
+; GFX9-NEXT: v_mov_b32_e32 v1, s1
+; GFX9-NEXT: flat_load_short_d16 v2, v[0:1] offset:8
+; GFX9-NEXT: v_mov_b32_e32 v0, s2
+; GFX9-NEXT: v_mov_b32_e32 v1, s3
+; GFX9-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX9-NEXT: v_pk_add_u16 v2, v2, v2
+; GFX9-NEXT: flat_store_dword v[0:1], v2
+; GFX9-NEXT: s_endpgm
+;
+; GFX10-LABEL: load_i16_lo:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
+; GFX10-NEXT: v_mov_b32_e32 v2, 0
+; GFX10-NEXT: s_waitcnt lgkmcnt(0)
+; GFX10-NEXT: s_add_u32 s0, s0, 8
+; GFX10-NEXT: s_addc_u32 s1, s1, 0
+; GFX10-NEXT: v_mov_b32_e32 v0, s0
+; GFX10-NEXT: v_mov_b32_e32 v1, s1
+; GFX10-NEXT: flat_load_short_d16 v2, v[0:1]
+; GFX10-NEXT: v_mov_b32_e32 v0, s2
+; GFX10-NEXT: v_mov_b32_e32 v1, s3
+; GFX10-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX10-NEXT: v_pk_add_u16 v2, v2, v2
+; GFX10-NEXT: flat_store_dword v[0:1], v2
+; GFX10-NEXT: s_endpgm
+;
+; GFX11-LABEL: load_i16_lo:
+; GFX11: ; %bb.0:
+; GFX11-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
+; GFX11-NEXT: v_mov_b32_e32 v2, 0
+; GFX11-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
+; GFX11-NEXT: flat_load_d16_b16 v2, v[0:1] offset:8
+; GFX11-NEXT: v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, s3
+; GFX11-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX11-NEXT: v_pk_add_u16 v2, v2, v2
+; GFX11-NEXT: flat_store_b32 v[0:1], v2
+; GFX11-NEXT: s_endpgm
%gep = getelementptr inbounds i16, ptr %arg, i32 4
%ld = load i16, ptr %gep, align 2
%vec = insertelement <2 x i16> <i16 poison, i16 0>, i16 %ld, i32 0
@@ -38,10 +130,51 @@ define amdgpu_kernel void @load_i16_lo(ptr %arg, ptr %out) {
ret void
}
-; GCN-LABEL: load_i16_hi:
-; GFX9_11: flat_load_{{short_d16_hi|d16_hi_b16}} v{{[0-9]+}}, v[{{[0-9:]+}}] offset:8{{$}}
-; GFX10: flat_load_short_d16_hi v{{[0-9]+}}, v[{{[0-9:]+}}]{{$}}
define amdgpu_kernel void @load_i16_hi(ptr %arg, ptr %out) {
+; GFX9-LABEL: load_i16_hi:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
+; GFX9-NEXT: v_mov_b32_e32 v2, 0
+; GFX9-NEXT: s_waitcnt lgkmcnt(0)
+; GFX9-NEXT: v_mov_b32_e32 v0, s0
+; GFX9-NEXT: v_mov_b32_e32 v1, s1
+; GFX9-NEXT: flat_load_short_d16_hi v2, v[0:1] offset:8
+; GFX9-NEXT: v_mov_b32_e32 v0, s2
+; GFX9-NEXT: v_mov_b32_e32 v1, s3
+; GFX9-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX9-NEXT: v_pk_add_u16 v2, v2, v2
+; GFX9-NEXT: flat_store_dword v[0:1], v2
+; GFX9-NEXT: s_endpgm
+;
+; GFX10-LABEL: load_i16_hi:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
+; GFX10-NEXT: v_mov_b32_e32 v2, 0
+; GFX10-NEXT: s_waitcnt lgkmcnt(0)
+; GFX10-NEXT: s_add_u32 s0, s0, 8
+; GFX10-NEXT: s_addc_u32 s1, s1, 0
+; GFX10-NEXT: v_mov_b32_e32 v0, s0
+; GFX10-NEXT: v_mov_b32_e32 v1, s1
+; GFX10-NEXT: flat_load_short_d16_hi v2, v[0:1]
+; GFX10-NEXT: v_mov_b32_e32 v0, s2
+; GFX10-NEXT: v_mov_b32_e32 v1, s3
+; GFX10-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX10-NEXT: v_pk_add_u16 v2, v2, v2
+; GFX10-NEXT: flat_store_dword v[0:1], v2
+; GFX10-NEXT: s_endpgm
+;
+; GFX11-LABEL: load_i16_hi:
+; GFX11: ; %bb.0:
+; GFX11-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
+; GFX11-NEXT: v_mov_b32_e32 v2, 0
+; GFX11-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
+; GFX11-NEXT: flat_load_d16_hi_b16 v2, v[0:1] offset:8
+; GFX11-NEXT: v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, s3
+; GFX11-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX11-NEXT: v_pk_add_u16 v2, v2, v2
+; GFX11-NEXT: flat_store_b32 v[0:1], v2
+; GFX11-NEXT: s_endpgm
%gep = getelementptr inbounds i16, ptr %arg, i32 4
%ld = load i16, ptr %gep, align 2
%vec = insertelement <2 x i16> <i16 0, i16 poison>, i16 %ld, i32 1
@@ -50,10 +183,51 @@ define amdgpu_kernel void @load_i16_hi(ptr %arg, ptr %out) {
ret void
}
-; GCN-LABEL: load_half_lo:
-; GFX9_11: flat_load_{{short_d16|d16_b16}} v{{[0-9]+}}, v[{{[0-9:]+}}] offset:8{{$}}
-; GFX10: flat_load_short_d16 v{{[0-9]+}}, v[{{[0-9:]+}}]{{$}}
define amdgpu_kernel void @load_half_lo(ptr %arg, ptr %out) {
+; GFX9-LABEL: load_half_lo:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
+; GFX9-NEXT: v_mov_b32_e32 v2, 0
+; GFX9-NEXT: s_waitcnt lgkmcnt(0)
+; GFX9-NEXT: v_mov_b32_e32 v0, s0
+; GFX9-NEXT: v_mov_b32_e32 v1, s1
+; GFX9-NEXT: flat_load_short_d16 v2, v[0:1] offset:8
+; GFX9-NEXT: v_mov_b32_e32 v0, s2
+; GFX9-NEXT: v_mov_b32_e32 v1, s3
+; GFX9-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX9-NEXT: v_pk_add_f16 v2, v2, v2
+; GFX9-NEXT: flat_store_dword v[0:1], v2
+; GFX9-NEXT: s_endpgm
+;
+; GFX10-LABEL: load_half_lo:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
+; GFX10-NEXT: v_mov_b32_e32 v2, 0
+; GFX10-NEXT: s_waitcnt lgkmcnt(0)
+; GFX10-NEXT: s_add_u32 s0, s0, 8
+; GFX10-NEXT: s_addc_u32 s1, s1, 0
+; GFX10-NEXT: v_mov_b32_e32 v0, s0
+; GFX10-NEXT: v_mov_b32_e32 v1, s1
+; GFX10-NEXT: flat_load_short_d16 v2, v[0:1]
+; GFX10-NEXT: v_mov_b32_e32 v0, s2
+; GFX10-NEXT: v_mov_b32_e32 v1, s3
+; GFX10-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX10-NEXT: v_pk_add_f16 v2, v2, v2
+; GFX10-NEXT: flat_store_dword v[0:1], v2
+; GFX10-NEXT: s_endpgm
+;
+; GFX11-LABEL: load_half_lo:
+; GFX11: ; %bb.0:
+; GFX11-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
+; GFX11-NEXT: v_mov_b32_e32 v2, 0
+; GFX11-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
+; GFX11-NEXT: flat_load_d16_b16 v2, v[0:1] offset:8
+; GFX11-NEXT: v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, s3
+; GFX11-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX11-NEXT: v_pk_add_f16 v2, v2, v2
+; GFX11-NEXT: flat_store_b32 v[0:1], v2
+; GFX11-NEXT: s_endpgm
%gep = getelementptr inbounds half, ptr %arg, i32 4
%ld = load half, ptr %gep, align 2
%vec = insertelement <2 x half> <half poison, half 0xH0000>, half %ld, i32 0
@@ -62,10 +236,51 @@ define amdgpu_kernel void @load_half_lo(ptr %arg, ptr %out) {
ret void
}
-; GCN-LABEL: load_half_hi:
-; GFX9_11: flat_load_{{short_d16_hi|d16_hi_b16}} v{{[0-9]+}}, v[{{[0-9:]+}}] offset:8{{$}}
-; GFX10: flat_load_short_d16_hi v{{[0-9]+}}, v[{{[0-9:]+}}]{{$}}
define amdgpu_kernel void @load_half_hi(ptr %arg, ptr %out) {
+; GFX9-LABEL: load_half_hi:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
+; GFX9-NEXT: v_mov_b32_e32 v2, 0
+; GFX9-NEXT: s_waitcnt lgkmcnt(0)
+; GFX9-NEXT: v_mov_b32_e32 v0, s0
+; GFX9-NEXT: v_mov_b32_e32 v1, s1
+; GFX9-NEXT: flat_load_short_d16_hi v2, v[0:1] offset:8
+; GFX9-NEXT: v_mov_b32_e32 v0, s2
+; GFX9-NEXT: v_mov_b32_e32 v1, s3
+; GFX9-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX9-NEXT: v_pk_add_f16 v2, v2, v2
+; GFX9-NEXT: flat_store_dword v[0:1], v2
+; GFX9-NEXT: s_endpgm
+;
+; GFX10-LABEL: load_half_hi:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
+; GFX10-NEXT: v_mov_b32_e32 v2, 0
+; GFX10-NEXT: s_waitcnt lgkmcnt(0)
+; GFX10-NEXT: s_add_u32 s0, s0, 8
+; GFX10-NEXT: s_addc_u32 s1, s1, 0
+; GFX10-NEXT: v_mov_b32_e32 v0, s0
+; GFX10-NEXT: v_mov_b32_e32 v1, s1
+; GFX10-NEXT: flat_load_short_d16_hi v2, v[0:1]
+; GFX10-NEXT: v_mov_b32_e32 v0, s2
+; GFX10-NEXT: v_mov_b32_e32 v1, s3
+; GFX10-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX10-NEXT: v_pk_add_f16 v2, v2, v2
+; GFX10-NEXT: flat_store_dword v[0:1], v2
+; GFX10-NEXT: s_endpgm
+;
+; GFX11-LABEL: load_half_hi:
+; GFX11: ; %bb.0:
+; GFX11-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
+; GFX11-NEXT: v_mov_b32_e32 v2, 0
+; GFX11-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
+; GFX11-NEXT: flat_load_d16_hi_b16 v2, v[0:1] offset:8
+; GFX11-NEXT: v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, s3
+; GFX11-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX11-NEXT: v_pk_add_f16 v2, v2, v2
+; GFX11-NEXT: flat_store_b32 v[0:1], v2
+; GFX11-NEXT: s_endpgm
%gep = getelementptr inbounds half, ptr %arg, i32 4
%ld = load half, ptr %gep, align 2
%vec = insertelement <2 x half> <half 0xH0000, half poison>, half %ld, i32 1
@@ -74,10 +289,48 @@ define amdgpu_kernel void @load_half_hi(ptr %arg, ptr %out) {
ret void
}
-; GCN-LABEL: load_float_lo:
-; GFX9_11: flat_load_{{dword|b32}} v{{[0-9]+}}, v[{{[0-9:]+}}] offset:16{{$}}
-; GFX10: flat_load_dword v{{[0-9]+}}, v[{{[0-9:]+}}]{{$}}
define amdgpu_kernel void @load_float_lo(ptr %arg, ptr %out) {
+; GFX9-LABEL: load_float_lo:
+; GFX9: ; %bb.0:
+; GFX9-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
+; GFX9-NEXT: s_waitcnt lgkmcnt(0)
+; GFX9-NEXT: v_mov_b32_e32 v0, s0
+; GFX9-NEXT: v_mov_b32_e32 v1, s1
+; GFX9-NEXT: flat_load_dword v2, v[0:1] offset:16
+; GFX9-NEXT: v_mov_b32_e32 v0, s2
+; GFX9-NEXT: v_mov_b32_e32 v1, s3
+; GFX9-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX9-NEXT: v_add_f32_e32 v2, v2, v2
+; GFX9-NEXT: flat_store_dword v[0:1], v2
+; GFX9-NEXT: s_endpgm
+;
+; GFX10-LABEL: load_float_lo:
+; GFX10: ; %bb.0:
+; GFX10-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
+; GFX10-NEXT: s_waitcnt lgkmcnt(0)
+; GFX10-NEXT: s_add_u32 s0, s0, 16
+; GFX10-NEXT: s_addc_u32 s1, s1, 0
+; GFX10-NEXT: v_mov_b32_e32 v0, s0
+; GFX10-NEXT: v_mov_b32_e32 v1, s1
+; GFX10-NEXT: flat_load_dword v2, v[0:1]
+; GFX10-NEXT: v_mov_b32_e32 v0, s2
+; GFX10-NEXT: v_mov_b32_e32 v1, s3
+; GFX10-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX10-NEXT: v_add_f32_e32 v2, v2, v2
+; GFX10-NEXT: flat_store_dword v[0:1], v2
+; GFX10-NEXT: s_endpgm
+;
+; GFX11-LABEL: load_float_lo:
+; GFX11: ; %bb.0:
+; GFX11-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
+; GFX11-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
+; GFX11-NEXT: flat_load_b32 v2, v[0:1] offset:16
+; GFX11-NEXT: v_dual_mov_b32 v0, s2 :: v_dual_mov_b32 v1, s3
+; GFX11-NEXT: s_waitcnt vmcnt(0) lgkmcnt(0)
+; GFX11-NEXT: v_add_f32_e32 v2, v2, v2
+; GFX11-NEXT: flat_store_b32 v[0:1], v2
+; GFX11-NEXT: s_endpgm
%gep = getelementptr inbounds float, ptr %arg, i32 4
%ld = load float, ptr %gep, align 4
%v = fadd float %ld, %ld
diff --git a/llvm/test/CodeGen/AMDGPU/lds-misaligned-bug.ll b/llvm/test/CodeGen/AMDGPU/lds-misaligned-bug.ll
index 278ad63b0b76c..7ffc2a6987742 100644
--- a/llvm/...
[truncated]
|
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5 | ||
; RUN: llc -mtriple=amdgcn -mcpu=bonaire -verify-machineinstrs < %s | FileCheck -enable-var-scope -check-prefix=CI %s | ||
; RUN: llc -mtriple=amdgcn -mcpu=tonga -mattr=-flat-for-global -verify-machineinstrs < %s | FileCheck -enable-var-scope -check-prefix=VI %s | ||
; RUN: llc -mtriple=amdgcn -mcpu=gfx900 -verify-machineinstrs < %s | FileCheck -enable-var-scope -check-prefix=GFX9 %s | ||
|
||
declare i32 @llvm.amdgcn.atomic.inc.i32.p1(ptr addrspace(1) nocapture, i32, i32, i32, i1) #2 |
There was a problem hiding this comment.
Choose a reason for hiding this comment
The reason will be displayed to describe this comment to others. Learn more.
I'm not sure why we still have these inc/dec tests, do we have equivalent coverage using atomicrmw uinc_wrap?
There was a problem hiding this comment.
Choose a reason for hiding this comment
The reason will be displayed to describe this comment to others. Learn more.
we do, such as llvm/test/CodeGen/AMDGPU/flat_atomics_i32_system.ll
, and llvm/test/CodeGen/AMDGPU/global_atomics_i32_system.ll
.
There was a problem hiding this comment.
Choose a reason for hiding this comment
The reason will be displayed to describe this comment to others. Learn more.
these specifically do not have system scope though
…137534) Make preparation for llvm#137488.
…137534) Make preparation for llvm#137488.
…137534) Make preparation for llvm#137488.
Make preparation for #137488.