1; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py 2; RUN: llc -global-isel -march=amdgcn -mcpu=gfx1010 -verify-machineinstrs < %s | FileCheck -check-prefixes=GFX10 %s 3; RUN: llc -global-isel -march=amdgcn -mcpu=gfx1100 -verify-machineinstrs < %s | FileCheck -check-prefixes=GFX10 %s 4 5define amdgpu_ps <4 x float> @sample_d_1d(<8 x i32> inreg %rsrc, <4 x i32> inreg %samp, half %dsdh, half %dsdv, float %s) { 6; GFX10-LABEL: sample_d_1d: 7; GFX10: ; %bb.0: ; %main_body 8; GFX10-NEXT: s_lshl_b32 s12, s0, 16 9; GFX10-NEXT: v_and_or_b32 v0, 0xffff, v0, s12 10; GFX10-NEXT: v_and_or_b32 v1, 0xffff, v1, s12 11; GFX10-NEXT: image_sample_d_g16 v[0:3], v[0:2], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_1D 12; GFX10-NEXT: s_waitcnt vmcnt(0) 13; GFX10-NEXT: ; return to shader part epilog 14main_body: 15 %v = call <4 x float> @llvm.amdgcn.image.sample.d.1d.v4f32.f16.f32(i32 15, half %dsdh, half %dsdv, float %s, <8 x i32> %rsrc, <4 x i32> %samp, i1 0, i32 0, i32 0) 16 ret <4 x float> %v 17} 18 19define amdgpu_ps <4 x float> @sample_d_2d(<8 x i32> inreg %rsrc, <4 x i32> inreg %samp, half %dsdh, half %dtdh, half %dsdv, half %dtdv, float %s, float %t) { 20; GFX10-LABEL: sample_d_2d: 21; GFX10: ; %bb.0: ; %main_body 22; GFX10-NEXT: v_lshlrev_b32_e32 v1, 16, v1 23; GFX10-NEXT: v_lshlrev_b32_e32 v3, 16, v3 24; GFX10-NEXT: v_and_or_b32 v0, 0xffff, v0, v1 25; GFX10-NEXT: v_and_or_b32 v1, 0xffff, v2, v3 26; GFX10-NEXT: image_sample_d_g16 v[0:3], [v0, v1, v4, v5], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_2D 27; GFX10-NEXT: s_waitcnt vmcnt(0) 28; GFX10-NEXT: ; return to shader part epilog 29main_body: 30 %v = call <4 x float> @llvm.amdgcn.image.sample.d.2d.v4f32.f16.f32(i32 15, half %dsdh, half %dtdh, half %dsdv, half %dtdv, float %s, float %t, <8 x i32> %rsrc, <4 x i32> %samp, i1 0, i32 0, i32 0) 31 ret <4 x float> %v 32} 33 34define amdgpu_ps <4 x float> @sample_d_3d(<8 x i32> inreg %rsrc, <4 x i32> inreg %samp, half %dsdh, half %dtdh, half %drdh, half %dsdv, half %dtdv, half %drdv, float %s, float %t, float %r) { 35; GFX10-LABEL: sample_d_3d: 36; GFX10: ; %bb.0: ; %main_body 37; GFX10-NEXT: v_mov_b32_e32 v9, v2 38; GFX10-NEXT: v_mov_b32_e32 v10, v3 39; GFX10-NEXT: v_lshlrev_b32_e32 v1, 16, v1 40; GFX10-NEXT: v_lshlrev_b32_e32 v4, 16, v4 41; GFX10-NEXT: s_lshl_b32 s12, s0, 16 42; GFX10-NEXT: v_and_or_b32 v3, 0xffff, v9, s12 43; GFX10-NEXT: v_and_or_b32 v2, 0xffff, v0, v1 44; GFX10-NEXT: v_and_or_b32 v4, 0xffff, v10, v4 45; GFX10-NEXT: v_and_or_b32 v5, 0xffff, v5, s12 46; GFX10-NEXT: image_sample_d_g16 v[0:3], v[2:8], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_3D 47; GFX10-NEXT: s_waitcnt vmcnt(0) 48; GFX10-NEXT: ; return to shader part epilog 49main_body: 50 %v = call <4 x float> @llvm.amdgcn.image.sample.d.3d.v4f32.f16.f32(i32 15, half %dsdh, half %dtdh, half %drdh, half %dsdv, half %dtdv, half %drdv, float %s, float %t, float %r, <8 x i32> %rsrc, <4 x i32> %samp, i1 0, i32 0, i32 0) 51 ret <4 x float> %v 52} 53 54define amdgpu_ps <4 x float> @sample_c_d_1d(<8 x i32> inreg %rsrc, <4 x i32> inreg %samp, float %zcompare, half %dsdh, half %dsdv, float %s) { 55; GFX10-LABEL: sample_c_d_1d: 56; GFX10: ; %bb.0: ; %main_body 57; GFX10-NEXT: s_lshl_b32 s12, s0, 16 58; GFX10-NEXT: v_and_or_b32 v1, 0xffff, v1, s12 59; GFX10-NEXT: v_and_or_b32 v2, 0xffff, v2, s12 60; GFX10-NEXT: image_sample_c_d_g16 v[0:3], v[0:3], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_1D 61; GFX10-NEXT: s_waitcnt vmcnt(0) 62; GFX10-NEXT: ; return to shader part epilog 63main_body: 64 %v = call <4 x float> @llvm.amdgcn.image.sample.c.d.1d.v4f32.f16.f32(i32 15, float %zcompare, half %dsdh, half %dsdv, float %s, <8 x i32> %rsrc, <4 x i32> %samp, i1 0, i32 0, i32 0) 65 ret <4 x float> %v 66} 67 68define amdgpu_ps <4 x float> @sample_c_d_2d(<8 x i32> inreg %rsrc, <4 x i32> inreg %samp, float %zcompare, half %dsdh, half %dtdh, half %dsdv, half %dtdv, float %s, float %t) { 69; GFX10-LABEL: sample_c_d_2d: 70; GFX10: ; %bb.0: ; %main_body 71; GFX10-NEXT: v_lshlrev_b32_e32 v2, 16, v2 72; GFX10-NEXT: v_lshlrev_b32_e32 v4, 16, v4 73; GFX10-NEXT: v_and_or_b32 v1, 0xffff, v1, v2 74; GFX10-NEXT: v_and_or_b32 v2, 0xffff, v3, v4 75; GFX10-NEXT: image_sample_c_d_g16 v[0:3], [v0, v1, v2, v5, v6], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_2D 76; GFX10-NEXT: s_waitcnt vmcnt(0) 77; GFX10-NEXT: ; return to shader part epilog 78main_body: 79 %v = call <4 x float> @llvm.amdgcn.image.sample.c.d.2d.v4f32.f16.f32(i32 15, float %zcompare, half %dsdh, half %dtdh, half %dsdv, half %dtdv, float %s, float %t, <8 x i32> %rsrc, <4 x i32> %samp, i1 0, i32 0, i32 0) 80 ret <4 x float> %v 81} 82 83define amdgpu_ps <4 x float> @sample_d_cl_1d(<8 x i32> inreg %rsrc, <4 x i32> inreg %samp, half %dsdh, half %dsdv, float %s, float %clamp) { 84; GFX10-LABEL: sample_d_cl_1d: 85; GFX10: ; %bb.0: ; %main_body 86; GFX10-NEXT: s_lshl_b32 s12, s0, 16 87; GFX10-NEXT: v_and_or_b32 v0, 0xffff, v0, s12 88; GFX10-NEXT: v_and_or_b32 v1, 0xffff, v1, s12 89; GFX10-NEXT: image_sample_d_cl_g16 v[0:3], v[0:3], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_1D 90; GFX10-NEXT: s_waitcnt vmcnt(0) 91; GFX10-NEXT: ; return to shader part epilog 92main_body: 93 %v = call <4 x float> @llvm.amdgcn.image.sample.d.cl.1d.v4f32.f16.f32(i32 15, half %dsdh, half %dsdv, float %s, float %clamp, <8 x i32> %rsrc, <4 x i32> %samp, i1 0, i32 0, i32 0) 94 ret <4 x float> %v 95} 96 97define amdgpu_ps <4 x float> @sample_d_cl_2d(<8 x i32> inreg %rsrc, <4 x i32> inreg %samp, half %dsdh, half %dtdh, half %dsdv, half %dtdv, float %s, float %t, float %clamp) { 98; GFX10-LABEL: sample_d_cl_2d: 99; GFX10: ; %bb.0: ; %main_body 100; GFX10-NEXT: v_lshlrev_b32_e32 v1, 16, v1 101; GFX10-NEXT: v_lshlrev_b32_e32 v3, 16, v3 102; GFX10-NEXT: v_and_or_b32 v0, 0xffff, v0, v1 103; GFX10-NEXT: v_and_or_b32 v1, 0xffff, v2, v3 104; GFX10-NEXT: image_sample_d_cl_g16 v[0:3], [v0, v1, v4, v5, v6], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_2D 105; GFX10-NEXT: s_waitcnt vmcnt(0) 106; GFX10-NEXT: ; return to shader part epilog 107main_body: 108 %v = call <4 x float> @llvm.amdgcn.image.sample.d.cl.2d.v4f32.f16.f32(i32 15, half %dsdh, half %dtdh, half %dsdv, half %dtdv, float %s, float %t, float %clamp, <8 x i32> %rsrc, <4 x i32> %samp, i1 0, i32 0, i32 0) 109 ret <4 x float> %v 110} 111 112define amdgpu_ps <4 x float> @sample_c_d_cl_1d(<8 x i32> inreg %rsrc, <4 x i32> inreg %samp, float %zcompare, half %dsdh, half %dsdv, float %s, float %clamp) { 113; GFX10-LABEL: sample_c_d_cl_1d: 114; GFX10: ; %bb.0: ; %main_body 115; GFX10-NEXT: s_lshl_b32 s12, s0, 16 116; GFX10-NEXT: v_and_or_b32 v1, 0xffff, v1, s12 117; GFX10-NEXT: v_and_or_b32 v2, 0xffff, v2, s12 118; GFX10-NEXT: image_sample_c_d_cl_g16 v[0:3], v[0:4], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_1D 119; GFX10-NEXT: s_waitcnt vmcnt(0) 120; GFX10-NEXT: ; return to shader part epilog 121main_body: 122 %v = call <4 x float> @llvm.amdgcn.image.sample.c.d.cl.1d.v4f32.f16.f32(i32 15, float %zcompare, half %dsdh, half %dsdv, float %s, float %clamp, <8 x i32> %rsrc, <4 x i32> %samp, i1 0, i32 0, i32 0) 123 ret <4 x float> %v 124} 125 126define amdgpu_ps <4 x float> @sample_c_d_cl_2d(<8 x i32> inreg %rsrc, <4 x i32> inreg %samp, float %zcompare, half %dsdh, half %dtdh, half %dsdv, half %dtdv, float %s, float %t, float %clamp) { 127; GFX10-LABEL: sample_c_d_cl_2d: 128; GFX10: ; %bb.0: ; %main_body 129; GFX10-NEXT: v_mov_b32_e32 v8, v2 130; GFX10-NEXT: v_mov_b32_e32 v9, v3 131; GFX10-NEXT: v_mov_b32_e32 v2, v0 132; GFX10-NEXT: v_lshlrev_b32_e32 v4, 16, v4 133; GFX10-NEXT: v_lshlrev_b32_e32 v0, 16, v8 134; GFX10-NEXT: v_and_or_b32 v4, 0xffff, v9, v4 135; GFX10-NEXT: v_and_or_b32 v3, 0xffff, v1, v0 136; GFX10-NEXT: image_sample_c_d_cl_g16 v[0:3], v[2:7], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_2D 137; GFX10-NEXT: s_waitcnt vmcnt(0) 138; GFX10-NEXT: ; return to shader part epilog 139main_body: 140 %v = call <4 x float> @llvm.amdgcn.image.sample.c.d.cl.2d.v4f32.f16.f32(i32 15, float %zcompare, half %dsdh, half %dtdh, half %dsdv, half %dtdv, float %s, float %t, float %clamp, <8 x i32> %rsrc, <4 x i32> %samp, i1 0, i32 0, i32 0) 141 ret <4 x float> %v 142} 143 144define amdgpu_ps float @sample_c_d_o_2darray_V1(<8 x i32> inreg %rsrc, <4 x i32> inreg %samp, i32 %offset, float %zcompare, half %dsdh, half %dtdh, half %dsdv, half %dtdv, float %s, float %t, float %slice) { 145; GFX10-LABEL: sample_c_d_o_2darray_V1: 146; GFX10: ; %bb.0: ; %main_body 147; GFX10-NEXT: v_mov_b32_e32 v9, v3 148; GFX10-NEXT: v_mov_b32_e32 v10, v2 149; GFX10-NEXT: v_mov_b32_e32 v11, v4 150; GFX10-NEXT: v_mov_b32_e32 v2, v0 151; GFX10-NEXT: v_mov_b32_e32 v3, v1 152; GFX10-NEXT: v_lshlrev_b32_e32 v0, 16, v9 153; GFX10-NEXT: v_lshlrev_b32_e32 v1, 16, v5 154; GFX10-NEXT: v_and_or_b32 v4, 0xffff, v10, v0 155; GFX10-NEXT: v_and_or_b32 v5, 0xffff, v11, v1 156; GFX10-NEXT: image_sample_c_d_o_g16 v0, v[2:8], s[0:7], s[8:11] dmask:0x4 dim:SQ_RSRC_IMG_2D_ARRAY 157; GFX10-NEXT: s_waitcnt vmcnt(0) 158; GFX10-NEXT: ; return to shader part epilog 159main_body: 160 %v = call float @llvm.amdgcn.image.sample.c.d.o.2darray.f16.f32.f32(i32 4, i32 %offset, float %zcompare, half %dsdh, half %dtdh, half %dsdv, half %dtdv, float %s, float %t, float %slice, <8 x i32> %rsrc, <4 x i32> %samp, i1 0, i32 0, i32 0) 161 ret float %v 162} 163 164define amdgpu_ps <2 x float> @sample_c_d_o_2darray_V2(<8 x i32> inreg %rsrc, <4 x i32> inreg %samp, i32 %offset, float %zcompare, half %dsdh, half %dtdh, half %dsdv, half %dtdv, float %s, float %t, float %slice) { 165; GFX10-LABEL: sample_c_d_o_2darray_V2: 166; GFX10: ; %bb.0: ; %main_body 167; GFX10-NEXT: v_mov_b32_e32 v9, v3 168; GFX10-NEXT: v_mov_b32_e32 v10, v2 169; GFX10-NEXT: v_mov_b32_e32 v11, v4 170; GFX10-NEXT: v_mov_b32_e32 v2, v0 171; GFX10-NEXT: v_mov_b32_e32 v3, v1 172; GFX10-NEXT: v_lshlrev_b32_e32 v0, 16, v9 173; GFX10-NEXT: v_lshlrev_b32_e32 v1, 16, v5 174; GFX10-NEXT: v_and_or_b32 v4, 0xffff, v10, v0 175; GFX10-NEXT: v_and_or_b32 v5, 0xffff, v11, v1 176; GFX10-NEXT: image_sample_c_d_o_g16 v[0:1], v[2:8], s[0:7], s[8:11] dmask:0x6 dim:SQ_RSRC_IMG_2D_ARRAY 177; GFX10-NEXT: s_waitcnt vmcnt(0) 178; GFX10-NEXT: ; return to shader part epilog 179main_body: 180 %v = call <2 x float> @llvm.amdgcn.image.sample.c.d.o.2darray.v2f32.f16.f32(i32 6, i32 %offset, float %zcompare, half %dsdh, half %dtdh, half %dsdv, half %dtdv, float %s, float %t, float %slice, <8 x i32> %rsrc, <4 x i32> %samp, i1 0, i32 0, i32 0) 181 ret <2 x float> %v 182} 183 184declare <4 x float> @llvm.amdgcn.image.sample.d.1d.v4f32.f16.f32(i32, half, half, float, <8 x i32>, <4 x i32>, i1, i32, i32) #1 185declare <4 x float> @llvm.amdgcn.image.sample.d.2d.v4f32.f16.f32(i32, half, half, half, half, float, float, <8 x i32>, <4 x i32>, i1, i32, i32) #1 186declare <4 x float> @llvm.amdgcn.image.sample.d.3d.v4f32.f16.f32(i32, half, half, half, half, half, half, float, float, float, <8 x i32>, <4 x i32>, i1, i32, i32) #1 187declare <4 x float> @llvm.amdgcn.image.sample.c.d.1d.v4f32.f16.f32(i32, float, half, half, float, <8 x i32>, <4 x i32>, i1, i32, i32) #1 188declare <4 x float> @llvm.amdgcn.image.sample.c.d.2d.v4f32.f16.f32(i32, float, half, half, half, half, float, float, <8 x i32>, <4 x i32>, i1, i32, i32) #1 189declare <4 x float> @llvm.amdgcn.image.sample.d.cl.1d.v4f32.f16.f32(i32, half, half, float, float, <8 x i32>, <4 x i32>, i1, i32, i32) #1 190declare <4 x float> @llvm.amdgcn.image.sample.d.cl.2d.v4f32.f16.f32(i32, half, half, half, half, float, float, float, <8 x i32>, <4 x i32>, i1, i32, i32) #1 191declare <4 x float> @llvm.amdgcn.image.sample.c.d.cl.1d.v4f32.f16.f32(i32, float, half, half, float, float, <8 x i32>, <4 x i32>, i1, i32, i32) #1 192declare <4 x float> @llvm.amdgcn.image.sample.c.d.cl.2d.v4f32.f16.f32(i32, float, half, half, half, half, float, float, float, <8 x i32>, <4 x i32>, i1, i32, i32) #1 193 194declare float @llvm.amdgcn.image.sample.c.d.o.2darray.f16.f32.f32(i32, i32, float, half, half, half, half, float, float, float, <8 x i32>, <4 x i32>, i1, i32, i32) #1 195declare <2 x float> @llvm.amdgcn.image.sample.c.d.o.2darray.v2f32.f16.f32(i32, i32, float, half, half, half, half, float, float, float, <8 x i32>, <4 x i32>, i1, i32, i32) #1 196 197attributes #0 = { nounwind } 198attributes #1 = { nounwind readonly } 199attributes #2 = { nounwind readnone } 200