llvm-6502/test/CodeGen/R600/array-ptr-calc-i32.ll

; RUN: llc -verify-machineinstrs -march=r600 -mcpu=SI -mattr=-promote-alloca < %s | FileCheck -check-prefix=SI-ALLOCA -check-prefix=SI %s
; RUN: llc -verify-machineinstrs -march=r600 -mcpu=SI -mattr=+promote-alloca < %s | FileCheck -check-prefix=SI-PROMOTE -check-prefix=SI %s

declare i32 @llvm.SI.tid() nounwind readnone
declare void @llvm.AMDGPU.barrier.local() nounwind noduplicate

; The required pointer calculations for the alloca'd actually requires
; an add and won't be folded into the addressing, which fails with a
; 64-bit pointer add. This should work since private pointers should
; be 32-bits.

; SI-LABEL: @test_private_array_ptr_calc:

; FIXME: We end up with zero argument for ADD, because
; SIRegisterInfo::eliminateFrameIndex() blindly replaces the frame index
; with the appropriate offset.  We should fold this into the store.
; SI-ALLOCA: V_ADD_I32_e32 [[PTRREG:v[0-9]+]], 0, v{{[0-9]+}}
; SI-ALLOCA: BUFFER_STORE_DWORD {{v[0-9]+}}, s[{{[0-9]+:[0-9]+}}], [[PTRREG]]
;
; FIXME: The AMDGPUPromoteAlloca pass should be able to convert this
; alloca to a vector.  It currently fails because it does not know how
; to interpret:
; getelementptr [4 x i32]* %alloca, i32 1, i32 %b

; SI-PROMOTE: V_ADD_I32_e32 [[PTRREG:v[0-9]+]]
; SI-PROMOTE: DS_WRITE_B32 {{v[0-9]+}}, [[PTRREG]]
define void @test_private_array_ptr_calc(i32 addrspace(1)* noalias %out, i32 addrspace(1)* noalias %inA, i32 addrspace(1)* noalias %inB) {
  %alloca = alloca [4 x i32], i32 4, align 16
  %tid = call i32 @llvm.SI.tid() readnone
  %a_ptr = getelementptr i32 addrspace(1)* %inA, i32 %tid
  %b_ptr = getelementptr i32 addrspace(1)* %inB, i32 %tid
  %a = load i32 addrspace(1)* %a_ptr
  %b = load i32 addrspace(1)* %b_ptr
  %result = add i32 %a, %b
  %alloca_ptr = getelementptr [4 x i32]* %alloca, i32 1, i32 %b
  store i32 %result, i32* %alloca_ptr, align 4
  ; Dummy call
  call void @llvm.AMDGPU.barrier.local() nounwind noduplicate
  %reload = load i32* %alloca_ptr, align 4
  %out_ptr = getelementptr i32 addrspace(1)* %out, i32 %tid
  store i32 %reload, i32 addrspace(1)* %out_ptr, align 4
  ret void
}
R600: Run more tests with promote alloca disabled. Re-run tests changed in r211110 to test both paths. Also fix broken check line. git-svn-id: https://llvm.org/svn/llvm-project/llvm/trunk@212895 91177308-0d34-0410-b5e6-96231b3b80d8 2014-07-13 02:46:17 +00:00			`; RUN: llc -verify-machineinstrs -march=r600 -mcpu=SI -mattr=-promote-alloca < %s \| FileCheck -check-prefix=SI-ALLOCA -check-prefix=SI %s`
			`; RUN: llc -verify-machineinstrs -march=r600 -mcpu=SI -mattr=+promote-alloca < %s \| FileCheck -check-prefix=SI-PROMOTE -check-prefix=SI %s`
R600/SI: Make private pointers be 32-bit. Different sized address spaces should theoretically work most of the time now, and since 64-bit add is currently disabled, using more 32-bit pointers fixes some cases. git-svn-id: https://llvm.org/svn/llvm-project/llvm/trunk@197659 91177308-0d34-0410-b5e6-96231b3b80d8 2013-12-19 05:32:55 +00:00
			`declare i32 @llvm.SI.tid() nounwind readnone`
			`declare void @llvm.AMDGPU.barrier.local() nounwind noduplicate`

			`; The required pointer calculations for the alloca'd actually requires`
			`; an add and won't be folded into the addressing, which fails with a`
			`; 64-bit pointer add. This should work since private pointers should`
			`; be 32-bits.`

			`; SI-LABEL: @test_private_array_ptr_calc:`
R600: Run more tests with promote alloca disabled. Re-run tests changed in r211110 to test both paths. Also fix broken check line. git-svn-id: https://llvm.org/svn/llvm-project/llvm/trunk@212895 91177308-0d34-0410-b5e6-96231b3b80d8 2014-07-13 02:46:17 +00:00
R600/SI: Use scratch memory for large private arrays git-svn-id: https://llvm.org/svn/llvm-project/llvm/trunk@213551 91177308-0d34-0410-b5e6-96231b3b80d8 2014-07-21 15:45:01 +00:00			`; FIXME: We end up with zero argument for ADD, because`
			`; SIRegisterInfo::eliminateFrameIndex() blindly replaces the frame index`
			`; with the appropriate offset. We should fold this into the store.`
			`; SI-ALLOCA: V_ADD_I32_e32 [[PTRREG:v[0-9]+]], 0, v{{[0-9]+}}`
			`; SI-ALLOCA: BUFFER_STORE_DWORD {{v[0-9]+}}, s[{{[0-9]+:[0-9]+}}], [[PTRREG]]`
R600: Use LDS and vectors for private memory git-svn-id: https://llvm.org/svn/llvm-project/llvm/trunk@211110 91177308-0d34-0410-b5e6-96231b3b80d8 2014-06-17 16:53:14 +00:00			`;`
			`; FIXME: The AMDGPUPromoteAlloca pass should be able to convert this`
			`; alloca to a vector. It currently fails because it does not know how`
			`; to interpret:`
			`; getelementptr [4 x i32]* %alloca, i32 1, i32 %b`
R600: Run more tests with promote alloca disabled. Re-run tests changed in r211110 to test both paths. Also fix broken check line. git-svn-id: https://llvm.org/svn/llvm-project/llvm/trunk@212895 91177308-0d34-0410-b5e6-96231b3b80d8 2014-07-13 02:46:17 +00:00
R600/SI: Use scratch memory for large private arrays git-svn-id: https://llvm.org/svn/llvm-project/llvm/trunk@213551 91177308-0d34-0410-b5e6-96231b3b80d8 2014-07-21 15:45:01 +00:00			`; SI-PROMOTE: V_ADD_I32_e32 [[PTRREG:v[0-9]+]]`
R600: Run more tests with promote alloca disabled. Re-run tests changed in r211110 to test both paths. Also fix broken check line. git-svn-id: https://llvm.org/svn/llvm-project/llvm/trunk@212895 91177308-0d34-0410-b5e6-96231b3b80d8 2014-07-13 02:46:17 +00:00			`; SI-PROMOTE: DS_WRITE_B32 {{v[0-9]+}}, [[PTRREG]]`
R600/SI: Make private pointers be 32-bit. Different sized address spaces should theoretically work most of the time now, and since 64-bit add is currently disabled, using more 32-bit pointers fixes some cases. git-svn-id: https://llvm.org/svn/llvm-project/llvm/trunk@197659 91177308-0d34-0410-b5e6-96231b3b80d8 2013-12-19 05:32:55 +00:00			`define void @test_private_array_ptr_calc(i32 addrspace(1)* noalias %out, i32 addrspace(1)* noalias %inA, i32 addrspace(1)* noalias %inB) {`
			`%alloca = alloca [4 x i32], i32 4, align 16`
			`%tid = call i32 @llvm.SI.tid() readnone`
			`%a_ptr = getelementptr i32 addrspace(1)* %inA, i32 %tid`
			`%b_ptr = getelementptr i32 addrspace(1)* %inB, i32 %tid`
			`%a = load i32 addrspace(1)* %a_ptr`
			`%b = load i32 addrspace(1)* %b_ptr`
			`%result = add i32 %a, %b`
			`%alloca_ptr = getelementptr [4 x i32]* %alloca, i32 1, i32 %b`
			`store i32 %result, i32* %alloca_ptr, align 4`
			`; Dummy call`
			`call void @llvm.AMDGPU.barrier.local() nounwind noduplicate`
			`%reload = load i32* %alloca_ptr, align 4`
			`%out_ptr = getelementptr i32 addrspace(1)* %out, i32 %tid`
			`store i32 %reload, i32 addrspace(1)* %out_ptr, align 4`
			`ret void`
			`}`