1; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
2; RUN: llc -global-isel -march=amdgcn -mcpu=gfx1010 -verify-machineinstrs < %s | FileCheck -check-prefixes=GFX10 %s
3; RUN: llc -global-isel -march=amdgcn -mcpu=gfx1100 -verify-machineinstrs < %s | FileCheck -check-prefixes=GFX10 %s
4
5define amdgpu_ps <4 x float> @sample_d_1d(<8 x i32> inreg %rsrc, <4 x i32> inreg %samp, half %dsdh, half %dsdv, float %s) {
6; GFX10-LABEL: sample_d_1d:
7; GFX10:       ; %bb.0: ; %main_body
8; GFX10-NEXT:    s_lshl_b32 s12, s0, 16
9; GFX10-NEXT:    v_and_or_b32 v0, 0xffff, v0, s12
10; GFX10-NEXT:    v_and_or_b32 v1, 0xffff, v1, s12
11; GFX10-NEXT:    image_sample_d_g16 v[0:3], v[0:2], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_1D
12; GFX10-NEXT:    s_waitcnt vmcnt(0)
13; GFX10-NEXT:    ; return to shader part epilog
14main_body:
15  %v = call <4 x float> @llvm.amdgcn.image.sample.d.1d.v4f32.f16.f32(i32 15, half %dsdh, half %dsdv, float %s, <8 x i32> %rsrc, <4 x i32> %samp, i1 0, i32 0, i32 0)
16  ret <4 x float> %v
17}
18
19define amdgpu_ps <4 x float> @sample_d_2d(<8 x i32> inreg %rsrc, <4 x i32> inreg %samp, half %dsdh, half %dtdh, half %dsdv, half %dtdv, float %s, float %t) {
20; GFX10-LABEL: sample_d_2d:
21; GFX10:       ; %bb.0: ; %main_body
22; GFX10-NEXT:    v_lshlrev_b32_e32 v1, 16, v1
23; GFX10-NEXT:    v_lshlrev_b32_e32 v3, 16, v3
24; GFX10-NEXT:    v_and_or_b32 v0, 0xffff, v0, v1
25; GFX10-NEXT:    v_and_or_b32 v1, 0xffff, v2, v3
26; GFX10-NEXT:    image_sample_d_g16 v[0:3], [v0, v1, v4, v5], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_2D
27; GFX10-NEXT:    s_waitcnt vmcnt(0)
28; GFX10-NEXT:    ; return to shader part epilog
29main_body:
30  %v = call <4 x float> @llvm.amdgcn.image.sample.d.2d.v4f32.f16.f32(i32 15, half %dsdh, half %dtdh, half %dsdv, half %dtdv, float %s, float %t, <8 x i32> %rsrc, <4 x i32> %samp, i1 0, i32 0, i32 0)
31  ret <4 x float> %v
32}
33
34define amdgpu_ps <4 x float> @sample_d_3d(<8 x i32> inreg %rsrc, <4 x i32> inreg %samp, half %dsdh, half %dtdh, half %drdh, half %dsdv, half %dtdv, half %drdv, float %s, float %t, float %r) {
35; GFX10-LABEL: sample_d_3d:
36; GFX10:       ; %bb.0: ; %main_body
37; GFX10-NEXT:    v_mov_b32_e32 v9, v2
38; GFX10-NEXT:    v_mov_b32_e32 v10, v3
39; GFX10-NEXT:    v_lshlrev_b32_e32 v1, 16, v1
40; GFX10-NEXT:    v_lshlrev_b32_e32 v4, 16, v4
41; GFX10-NEXT:    s_lshl_b32 s12, s0, 16
42; GFX10-NEXT:    v_and_or_b32 v3, 0xffff, v9, s12
43; GFX10-NEXT:    v_and_or_b32 v2, 0xffff, v0, v1
44; GFX10-NEXT:    v_and_or_b32 v4, 0xffff, v10, v4
45; GFX10-NEXT:    v_and_or_b32 v5, 0xffff, v5, s12
46; GFX10-NEXT:    image_sample_d_g16 v[0:3], v[2:8], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_3D
47; GFX10-NEXT:    s_waitcnt vmcnt(0)
48; GFX10-NEXT:    ; return to shader part epilog
49main_body:
50  %v = call <4 x float> @llvm.amdgcn.image.sample.d.3d.v4f32.f16.f32(i32 15, half %dsdh, half %dtdh, half %drdh, half %dsdv, half %dtdv, half %drdv, float %s, float %t, float %r, <8 x i32> %rsrc, <4 x i32> %samp, i1 0, i32 0, i32 0)
51  ret <4 x float> %v
52}
53
54define amdgpu_ps <4 x float> @sample_c_d_1d(<8 x i32> inreg %rsrc, <4 x i32> inreg %samp, float %zcompare, half %dsdh, half %dsdv, float %s) {
55; GFX10-LABEL: sample_c_d_1d:
56; GFX10:       ; %bb.0: ; %main_body
57; GFX10-NEXT:    s_lshl_b32 s12, s0, 16
58; GFX10-NEXT:    v_and_or_b32 v1, 0xffff, v1, s12
59; GFX10-NEXT:    v_and_or_b32 v2, 0xffff, v2, s12
60; GFX10-NEXT:    image_sample_c_d_g16 v[0:3], v[0:3], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_1D
61; GFX10-NEXT:    s_waitcnt vmcnt(0)
62; GFX10-NEXT:    ; return to shader part epilog
63main_body:
64  %v = call <4 x float> @llvm.amdgcn.image.sample.c.d.1d.v4f32.f16.f32(i32 15, float %zcompare, half %dsdh, half %dsdv, float %s, <8 x i32> %rsrc, <4 x i32> %samp, i1 0, i32 0, i32 0)
65  ret <4 x float> %v
66}
67
68define amdgpu_ps <4 x float> @sample_c_d_2d(<8 x i32> inreg %rsrc, <4 x i32> inreg %samp, float %zcompare, half %dsdh, half %dtdh, half %dsdv, half %dtdv, float %s, float %t) {
69; GFX10-LABEL: sample_c_d_2d:
70; GFX10:       ; %bb.0: ; %main_body
71; GFX10-NEXT:    v_lshlrev_b32_e32 v2, 16, v2
72; GFX10-NEXT:    v_lshlrev_b32_e32 v4, 16, v4
73; GFX10-NEXT:    v_and_or_b32 v1, 0xffff, v1, v2
74; GFX10-NEXT:    v_and_or_b32 v2, 0xffff, v3, v4
75; GFX10-NEXT:    image_sample_c_d_g16 v[0:3], [v0, v1, v2, v5, v6], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_2D
76; GFX10-NEXT:    s_waitcnt vmcnt(0)
77; GFX10-NEXT:    ; return to shader part epilog
78main_body:
79  %v = call <4 x float> @llvm.amdgcn.image.sample.c.d.2d.v4f32.f16.f32(i32 15, float %zcompare, half %dsdh, half %dtdh, half %dsdv, half %dtdv, float %s, float %t, <8 x i32> %rsrc, <4 x i32> %samp, i1 0, i32 0, i32 0)
80  ret <4 x float> %v
81}
82
83define amdgpu_ps <4 x float> @sample_d_cl_1d(<8 x i32> inreg %rsrc, <4 x i32> inreg %samp, half %dsdh, half %dsdv, float %s, float %clamp) {
84; GFX10-LABEL: sample_d_cl_1d:
85; GFX10:       ; %bb.0: ; %main_body
86; GFX10-NEXT:    s_lshl_b32 s12, s0, 16
87; GFX10-NEXT:    v_and_or_b32 v0, 0xffff, v0, s12
88; GFX10-NEXT:    v_and_or_b32 v1, 0xffff, v1, s12
89; GFX10-NEXT:    image_sample_d_cl_g16 v[0:3], v[0:3], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_1D
90; GFX10-NEXT:    s_waitcnt vmcnt(0)
91; GFX10-NEXT:    ; return to shader part epilog
92main_body:
93  %v = call <4 x float> @llvm.amdgcn.image.sample.d.cl.1d.v4f32.f16.f32(i32 15, half %dsdh, half %dsdv, float %s, float %clamp, <8 x i32> %rsrc, <4 x i32> %samp, i1 0, i32 0, i32 0)
94  ret <4 x float> %v
95}
96
97define amdgpu_ps <4 x float> @sample_d_cl_2d(<8 x i32> inreg %rsrc, <4 x i32> inreg %samp, half %dsdh, half %dtdh, half %dsdv, half %dtdv, float %s, float %t, float %clamp) {
98; GFX10-LABEL: sample_d_cl_2d:
99; GFX10:       ; %bb.0: ; %main_body
100; GFX10-NEXT:    v_lshlrev_b32_e32 v1, 16, v1
101; GFX10-NEXT:    v_lshlrev_b32_e32 v3, 16, v3
102; GFX10-NEXT:    v_and_or_b32 v0, 0xffff, v0, v1
103; GFX10-NEXT:    v_and_or_b32 v1, 0xffff, v2, v3
104; GFX10-NEXT:    image_sample_d_cl_g16 v[0:3], [v0, v1, v4, v5, v6], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_2D
105; GFX10-NEXT:    s_waitcnt vmcnt(0)
106; GFX10-NEXT:    ; return to shader part epilog
107main_body:
108  %v = call <4 x float> @llvm.amdgcn.image.sample.d.cl.2d.v4f32.f16.f32(i32 15, half %dsdh, half %dtdh, half %dsdv, half %dtdv, float %s, float %t, float %clamp, <8 x i32> %rsrc, <4 x i32> %samp, i1 0, i32 0, i32 0)
109  ret <4 x float> %v
110}
111
112define amdgpu_ps <4 x float> @sample_c_d_cl_1d(<8 x i32> inreg %rsrc, <4 x i32> inreg %samp, float %zcompare, half %dsdh, half %dsdv, float %s, float %clamp) {
113; GFX10-LABEL: sample_c_d_cl_1d:
114; GFX10:       ; %bb.0: ; %main_body
115; GFX10-NEXT:    s_lshl_b32 s12, s0, 16
116; GFX10-NEXT:    v_and_or_b32 v1, 0xffff, v1, s12
117; GFX10-NEXT:    v_and_or_b32 v2, 0xffff, v2, s12
118; GFX10-NEXT:    image_sample_c_d_cl_g16 v[0:3], v[0:4], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_1D
119; GFX10-NEXT:    s_waitcnt vmcnt(0)
120; GFX10-NEXT:    ; return to shader part epilog
121main_body:
122  %v = call <4 x float> @llvm.amdgcn.image.sample.c.d.cl.1d.v4f32.f16.f32(i32 15, float %zcompare, half %dsdh, half %dsdv, float %s, float %clamp, <8 x i32> %rsrc, <4 x i32> %samp, i1 0, i32 0, i32 0)
123  ret <4 x float> %v
124}
125
126define amdgpu_ps <4 x float> @sample_c_d_cl_2d(<8 x i32> inreg %rsrc, <4 x i32> inreg %samp, float %zcompare, half %dsdh, half %dtdh, half %dsdv, half %dtdv, float %s, float %t, float %clamp) {
127; GFX10-LABEL: sample_c_d_cl_2d:
128; GFX10:       ; %bb.0: ; %main_body
129; GFX10-NEXT:    v_mov_b32_e32 v8, v2
130; GFX10-NEXT:    v_mov_b32_e32 v9, v3
131; GFX10-NEXT:    v_mov_b32_e32 v2, v0
132; GFX10-NEXT:    v_lshlrev_b32_e32 v4, 16, v4
133; GFX10-NEXT:    v_lshlrev_b32_e32 v0, 16, v8
134; GFX10-NEXT:    v_and_or_b32 v4, 0xffff, v9, v4
135; GFX10-NEXT:    v_and_or_b32 v3, 0xffff, v1, v0
136; GFX10-NEXT:    image_sample_c_d_cl_g16 v[0:3], v[2:7], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_2D
137; GFX10-NEXT:    s_waitcnt vmcnt(0)
138; GFX10-NEXT:    ; return to shader part epilog
139main_body:
140  %v = call <4 x float> @llvm.amdgcn.image.sample.c.d.cl.2d.v4f32.f16.f32(i32 15, float %zcompare, half %dsdh, half %dtdh, half %dsdv, half %dtdv, float %s, float %t, float %clamp, <8 x i32> %rsrc, <4 x i32> %samp, i1 0, i32 0, i32 0)
141  ret <4 x float> %v
142}
143
144define amdgpu_ps float @sample_c_d_o_2darray_V1(<8 x i32> inreg %rsrc, <4 x i32> inreg %samp, i32 %offset, float %zcompare, half %dsdh, half %dtdh, half %dsdv, half %dtdv, float %s, float %t, float %slice) {
145; GFX10-LABEL: sample_c_d_o_2darray_V1:
146; GFX10:       ; %bb.0: ; %main_body
147; GFX10-NEXT:    v_mov_b32_e32 v9, v3
148; GFX10-NEXT:    v_mov_b32_e32 v10, v2
149; GFX10-NEXT:    v_mov_b32_e32 v11, v4
150; GFX10-NEXT:    v_mov_b32_e32 v2, v0
151; GFX10-NEXT:    v_mov_b32_e32 v3, v1
152; GFX10-NEXT:    v_lshlrev_b32_e32 v0, 16, v9
153; GFX10-NEXT:    v_lshlrev_b32_e32 v1, 16, v5
154; GFX10-NEXT:    v_and_or_b32 v4, 0xffff, v10, v0
155; GFX10-NEXT:    v_and_or_b32 v5, 0xffff, v11, v1
156; GFX10-NEXT:    image_sample_c_d_o_g16 v0, v[2:8], s[0:7], s[8:11] dmask:0x4 dim:SQ_RSRC_IMG_2D_ARRAY
157; GFX10-NEXT:    s_waitcnt vmcnt(0)
158; GFX10-NEXT:    ; return to shader part epilog
159main_body:
160  %v = call float @llvm.amdgcn.image.sample.c.d.o.2darray.f16.f32.f32(i32 4, i32 %offset, float %zcompare, half %dsdh, half %dtdh, half %dsdv, half %dtdv, float %s, float %t, float %slice, <8 x i32> %rsrc, <4 x i32> %samp, i1 0, i32 0, i32 0)
161  ret float %v
162}
163
164define amdgpu_ps <2 x float> @sample_c_d_o_2darray_V2(<8 x i32> inreg %rsrc, <4 x i32> inreg %samp, i32 %offset, float %zcompare, half %dsdh, half %dtdh, half %dsdv, half %dtdv, float %s, float %t, float %slice) {
165; GFX10-LABEL: sample_c_d_o_2darray_V2:
166; GFX10:       ; %bb.0: ; %main_body
167; GFX10-NEXT:    v_mov_b32_e32 v9, v3
168; GFX10-NEXT:    v_mov_b32_e32 v10, v2
169; GFX10-NEXT:    v_mov_b32_e32 v11, v4
170; GFX10-NEXT:    v_mov_b32_e32 v2, v0
171; GFX10-NEXT:    v_mov_b32_e32 v3, v1
172; GFX10-NEXT:    v_lshlrev_b32_e32 v0, 16, v9
173; GFX10-NEXT:    v_lshlrev_b32_e32 v1, 16, v5
174; GFX10-NEXT:    v_and_or_b32 v4, 0xffff, v10, v0
175; GFX10-NEXT:    v_and_or_b32 v5, 0xffff, v11, v1
176; GFX10-NEXT:    image_sample_c_d_o_g16 v[0:1], v[2:8], s[0:7], s[8:11] dmask:0x6 dim:SQ_RSRC_IMG_2D_ARRAY
177; GFX10-NEXT:    s_waitcnt vmcnt(0)
178; GFX10-NEXT:    ; return to shader part epilog
179main_body:
180  %v = call <2 x float> @llvm.amdgcn.image.sample.c.d.o.2darray.v2f32.f16.f32(i32 6, i32 %offset, float %zcompare, half %dsdh, half %dtdh, half %dsdv, half %dtdv, float %s, float %t, float %slice, <8 x i32> %rsrc, <4 x i32> %samp, i1 0, i32 0, i32 0)
181  ret <2 x float> %v
182}
183
184declare <4 x float> @llvm.amdgcn.image.sample.d.1d.v4f32.f16.f32(i32, half, half, float, <8 x i32>, <4 x i32>, i1, i32, i32) #1
185declare <4 x float> @llvm.amdgcn.image.sample.d.2d.v4f32.f16.f32(i32, half, half, half, half, float, float, <8 x i32>, <4 x i32>, i1, i32, i32) #1
186declare <4 x float> @llvm.amdgcn.image.sample.d.3d.v4f32.f16.f32(i32, half, half, half, half, half, half, float, float, float, <8 x i32>, <4 x i32>, i1, i32, i32) #1
187declare <4 x float> @llvm.amdgcn.image.sample.c.d.1d.v4f32.f16.f32(i32, float, half, half, float, <8 x i32>, <4 x i32>, i1, i32, i32) #1
188declare <4 x float> @llvm.amdgcn.image.sample.c.d.2d.v4f32.f16.f32(i32, float, half, half, half, half, float, float, <8 x i32>, <4 x i32>, i1, i32, i32) #1
189declare <4 x float> @llvm.amdgcn.image.sample.d.cl.1d.v4f32.f16.f32(i32, half, half, float, float, <8 x i32>, <4 x i32>, i1, i32, i32) #1
190declare <4 x float> @llvm.amdgcn.image.sample.d.cl.2d.v4f32.f16.f32(i32, half, half, half, half, float, float, float, <8 x i32>, <4 x i32>, i1, i32, i32) #1
191declare <4 x float> @llvm.amdgcn.image.sample.c.d.cl.1d.v4f32.f16.f32(i32, float, half, half, float, float, <8 x i32>, <4 x i32>, i1, i32, i32) #1
192declare <4 x float> @llvm.amdgcn.image.sample.c.d.cl.2d.v4f32.f16.f32(i32, float, half, half, half, half, float, float, float, <8 x i32>, <4 x i32>, i1, i32, i32) #1
193
194declare float @llvm.amdgcn.image.sample.c.d.o.2darray.f16.f32.f32(i32, i32, float, half, half, half, half, float, float, float, <8 x i32>, <4 x i32>, i1, i32, i32) #1
195declare <2 x float> @llvm.amdgcn.image.sample.c.d.o.2darray.v2f32.f16.f32(i32, i32, float, half, half, half, half, float, float, float, <8 x i32>, <4 x i32>, i1, i32, i32) #1
196
197attributes #0 = { nounwind }
198attributes #1 = { nounwind readonly }
199attributes #2 = { nounwind readnone }
200