1; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py 2; RUN: llc -global-isel -march=amdgcn -mcpu=gfx1010 -verify-machineinstrs < %s | FileCheck -check-prefix=GFX10 %s 3 4define amdgpu_ps <4 x float> @sample_cd_1d(<8 x i32> inreg %rsrc, <4 x i32> inreg %samp, half %dsdh, half %dsdv, float %s) { 5; GFX10-LABEL: sample_cd_1d: 6; GFX10: ; %bb.0: ; %main_body 7; GFX10-NEXT: s_lshl_b32 s12, s0, 16 8; GFX10-NEXT: v_and_or_b32 v0, 0xffff, v0, s12 9; GFX10-NEXT: v_and_or_b32 v1, 0xffff, v1, s12 10; GFX10-NEXT: image_sample_cd_g16 v[0:3], v[0:2], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_1D 11; GFX10-NEXT: s_waitcnt vmcnt(0) 12; GFX10-NEXT: ; return to shader part epilog 13main_body: 14 %v = call <4 x float> @llvm.amdgcn.image.sample.cd.1d.v4f32.f16.f32(i32 15, half %dsdh, half %dsdv, float %s, <8 x i32> %rsrc, <4 x i32> %samp, i1 0, i32 0, i32 0) 15 ret <4 x float> %v 16} 17 18define amdgpu_ps <4 x float> @sample_cd_2d(<8 x i32> inreg %rsrc, <4 x i32> inreg %samp, half %dsdh, half %dtdh, half %dsdv, half %dtdv, float %s, float %t) { 19; GFX10-LABEL: sample_cd_2d: 20; GFX10: ; %bb.0: ; %main_body 21; GFX10-NEXT: v_lshlrev_b32_e32 v1, 16, v1 22; GFX10-NEXT: v_lshlrev_b32_e32 v3, 16, v3 23; GFX10-NEXT: v_and_or_b32 v0, 0xffff, v0, v1 24; GFX10-NEXT: v_and_or_b32 v1, 0xffff, v2, v3 25; GFX10-NEXT: image_sample_cd_g16 v[0:3], [v0, v1, v4, v5], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_2D 26; GFX10-NEXT: s_waitcnt vmcnt(0) 27; GFX10-NEXT: ; return to shader part epilog 28main_body: 29 %v = call <4 x float> @llvm.amdgcn.image.sample.cd.2d.v4f32.f16.f32(i32 15, half %dsdh, half %dtdh, half %dsdv, half %dtdv, float %s, float %t, <8 x i32> %rsrc, <4 x i32> %samp, i1 0, i32 0, i32 0) 30 ret <4 x float> %v 31} 32 33define amdgpu_ps <4 x float> @sample_c_cd_1d(<8 x i32> inreg %rsrc, <4 x i32> inreg %samp, float %zcompare, half %dsdh, half %dsdv, float %s) { 34; GFX10-LABEL: sample_c_cd_1d: 35; GFX10: ; %bb.0: ; %main_body 36; GFX10-NEXT: s_lshl_b32 s12, s0, 16 37; GFX10-NEXT: v_and_or_b32 v1, 0xffff, v1, s12 38; GFX10-NEXT: v_and_or_b32 v2, 0xffff, v2, s12 39; GFX10-NEXT: image_sample_c_cd_g16 v[0:3], v[0:3], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_1D 40; GFX10-NEXT: s_waitcnt vmcnt(0) 41; GFX10-NEXT: ; return to shader part epilog 42main_body: 43 %v = call <4 x float> @llvm.amdgcn.image.sample.c.cd.1d.v4f32.f16.f32(i32 15, float %zcompare, half %dsdh, half %dsdv, float %s, <8 x i32> %rsrc, <4 x i32> %samp, i1 0, i32 0, i32 0) 44 ret <4 x float> %v 45} 46 47define amdgpu_ps <4 x float> @sample_c_cd_2d(<8 x i32> inreg %rsrc, <4 x i32> inreg %samp, float %zcompare, half %dsdh, half %dtdh, half %dsdv, half %dtdv, float %s, float %t) { 48; GFX10-LABEL: sample_c_cd_2d: 49; GFX10: ; %bb.0: ; %main_body 50; GFX10-NEXT: v_lshlrev_b32_e32 v2, 16, v2 51; GFX10-NEXT: v_lshlrev_b32_e32 v4, 16, v4 52; GFX10-NEXT: v_and_or_b32 v1, 0xffff, v1, v2 53; GFX10-NEXT: v_and_or_b32 v2, 0xffff, v3, v4 54; GFX10-NEXT: image_sample_c_cd_g16 v[0:3], [v0, v1, v2, v5, v6], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_2D 55; GFX10-NEXT: s_waitcnt vmcnt(0) 56; GFX10-NEXT: ; return to shader part epilog 57main_body: 58 %v = call <4 x float> @llvm.amdgcn.image.sample.c.cd.2d.v4f32.f16.f32(i32 15, float %zcompare, half %dsdh, half %dtdh, half %dsdv, half %dtdv, float %s, float %t, <8 x i32> %rsrc, <4 x i32> %samp, i1 0, i32 0, i32 0) 59 ret <4 x float> %v 60} 61 62define amdgpu_ps <4 x float> @sample_cd_cl_1d(<8 x i32> inreg %rsrc, <4 x i32> inreg %samp, half %dsdh, half %dsdv, float %s, float %clamp) { 63; GFX10-LABEL: sample_cd_cl_1d: 64; GFX10: ; %bb.0: ; %main_body 65; GFX10-NEXT: s_lshl_b32 s12, s0, 16 66; GFX10-NEXT: v_and_or_b32 v0, 0xffff, v0, s12 67; GFX10-NEXT: v_and_or_b32 v1, 0xffff, v1, s12 68; GFX10-NEXT: image_sample_cd_cl_g16 v[0:3], v[0:3], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_1D 69; GFX10-NEXT: s_waitcnt vmcnt(0) 70; GFX10-NEXT: ; return to shader part epilog 71main_body: 72 %v = call <4 x float> @llvm.amdgcn.image.sample.cd.cl.1d.v4f32.f16.f32(i32 15, half %dsdh, half %dsdv, float %s, float %clamp, <8 x i32> %rsrc, <4 x i32> %samp, i1 0, i32 0, i32 0) 73 ret <4 x float> %v 74} 75 76define amdgpu_ps <4 x float> @sample_cd_cl_2d(<8 x i32> inreg %rsrc, <4 x i32> inreg %samp, half %dsdh, half %dtdh, half %dsdv, half %dtdv, float %s, float %t, float %clamp) { 77; GFX10-LABEL: sample_cd_cl_2d: 78; GFX10: ; %bb.0: ; %main_body 79; GFX10-NEXT: v_lshlrev_b32_e32 v1, 16, v1 80; GFX10-NEXT: v_lshlrev_b32_e32 v3, 16, v3 81; GFX10-NEXT: v_and_or_b32 v0, 0xffff, v0, v1 82; GFX10-NEXT: v_and_or_b32 v1, 0xffff, v2, v3 83; GFX10-NEXT: image_sample_cd_cl_g16 v[0:3], [v0, v1, v4, v5, v6], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_2D 84; GFX10-NEXT: s_waitcnt vmcnt(0) 85; GFX10-NEXT: ; return to shader part epilog 86main_body: 87 %v = call <4 x float> @llvm.amdgcn.image.sample.cd.cl.2d.v4f32.f16.f32(i32 15, half %dsdh, half %dtdh, half %dsdv, half %dtdv, float %s, float %t, float %clamp, <8 x i32> %rsrc, <4 x i32> %samp, i1 0, i32 0, i32 0) 88 ret <4 x float> %v 89} 90 91define amdgpu_ps <4 x float> @sample_c_cd_cl_1d(<8 x i32> inreg %rsrc, <4 x i32> inreg %samp, float %zcompare, half %dsdh, half %dsdv, float %s, float %clamp) { 92; GFX10-LABEL: sample_c_cd_cl_1d: 93; GFX10: ; %bb.0: ; %main_body 94; GFX10-NEXT: s_lshl_b32 s12, s0, 16 95; GFX10-NEXT: v_and_or_b32 v1, 0xffff, v1, s12 96; GFX10-NEXT: v_and_or_b32 v2, 0xffff, v2, s12 97; GFX10-NEXT: image_sample_c_cd_cl_g16 v[0:3], v[0:4], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_1D 98; GFX10-NEXT: s_waitcnt vmcnt(0) 99; GFX10-NEXT: ; return to shader part epilog 100main_body: 101 %v = call <4 x float> @llvm.amdgcn.image.sample.c.cd.cl.1d.v4f32.f16.f32(i32 15, float %zcompare, half %dsdh, half %dsdv, float %s, float %clamp, <8 x i32> %rsrc, <4 x i32> %samp, i1 0, i32 0, i32 0) 102 ret <4 x float> %v 103} 104 105define amdgpu_ps <4 x float> @sample_c_cd_cl_2d(<8 x i32> inreg %rsrc, <4 x i32> inreg %samp, float %zcompare, half %dsdh, half %dtdh, half %dsdv, half %dtdv, float %s, float %t, float %clamp) { 106; GFX10-LABEL: sample_c_cd_cl_2d: 107; GFX10: ; %bb.0: ; %main_body 108; GFX10-NEXT: v_mov_b32_e32 v8, v2 109; GFX10-NEXT: v_mov_b32_e32 v9, v3 110; GFX10-NEXT: v_mov_b32_e32 v2, v0 111; GFX10-NEXT: v_lshlrev_b32_e32 v4, 16, v4 112; GFX10-NEXT: v_lshlrev_b32_e32 v0, 16, v8 113; GFX10-NEXT: v_and_or_b32 v4, 0xffff, v9, v4 114; GFX10-NEXT: v_and_or_b32 v3, 0xffff, v1, v0 115; GFX10-NEXT: image_sample_c_cd_cl_g16 v[0:3], v[2:7], s[0:7], s[8:11] dmask:0xf dim:SQ_RSRC_IMG_2D 116; GFX10-NEXT: s_waitcnt vmcnt(0) 117; GFX10-NEXT: ; return to shader part epilog 118main_body: 119 %v = call <4 x float> @llvm.amdgcn.image.sample.c.cd.cl.2d.v4f32.f16.f32(i32 15, float %zcompare, half %dsdh, half %dtdh, half %dsdv, half %dtdv, float %s, float %t, float %clamp, <8 x i32> %rsrc, <4 x i32> %samp, i1 0, i32 0, i32 0) 120 ret <4 x float> %v 121} 122 123declare <4 x float> @llvm.amdgcn.image.sample.cd.1d.v4f32.f16.f32(i32, half, half, float, <8 x i32>, <4 x i32>, i1, i32, i32) #1 124declare <4 x float> @llvm.amdgcn.image.sample.cd.2d.v4f32.f16.f32(i32, half, half, half, half, float, float, <8 x i32>, <4 x i32>, i1, i32, i32) #1 125declare <4 x float> @llvm.amdgcn.image.sample.c.cd.1d.v4f32.f16.f32(i32, float, half, half, float, <8 x i32>, <4 x i32>, i1, i32, i32) #1 126declare <4 x float> @llvm.amdgcn.image.sample.c.cd.2d.v4f32.f16.f32(i32, float, half, half, half, half, float, float, <8 x i32>, <4 x i32>, i1, i32, i32) #1 127declare <4 x float> @llvm.amdgcn.image.sample.cd.cl.1d.v4f32.f16.f32(i32, half, half, float, float, <8 x i32>, <4 x i32>, i1, i32, i32) #1 128declare <4 x float> @llvm.amdgcn.image.sample.cd.cl.2d.v4f32.f16.f32(i32, half, half, half, half, float, float, float, <8 x i32>, <4 x i32>, i1, i32, i32) #1 129declare <4 x float> @llvm.amdgcn.image.sample.c.cd.cl.1d.v4f32.f16.f32(i32, float, half, half, float, float, <8 x i32>, <4 x i32>, i1, i32, i32) #1 130declare <4 x float> @llvm.amdgcn.image.sample.c.cd.cl.2d.v4f32.f16.f32(i32, float, half, half, half, half, float, float, float, <8 x i32>, <4 x i32>, i1, i32, i32) #1 131 132attributes #0 = { nounwind } 133attributes #1 = { nounwind readonly } 134attributes #2 = { nounwind readnone } 135