1  /* { dg-do run { target avx512fp16 } } */
       2  /* { dg-options "-O2 -mavx512fp16 -mavx512dq" } */
       3  
       4  #define AVX512FP16
       5  #include "avx512fp16-helper.h"
       6  
       7  #define N_ELEMS (AVX512F_LEN / 16)
       8  
       9  void NOINLINE
      10  EMULATE(sub_ph) (V512 * dest, V512 op1, V512 op2,
      11                  __mmask32 k, int zero_mask)
      12  {
      13      V512 v1, v2, v3, v4, v5, v6, v7, v8;
      14      int i;
      15      __mmask16 m1, m2;
      16  
      17      m1 = k & 0xffff;
      18      m2 = (k >> 16) & 0xffff;
      19  
      20      unpack_ph_2twops(op1, &v1, &v2);
      21      unpack_ph_2twops(op2, &v3, &v4);
      22      unpack_ph_2twops(*dest, &v7, &v8);
      23  
      24      for (i = 0; i < 16; i++) {
      25          if (((1 << i) & m1) == 0) {
      26              if (zero_mask) {
      27                 v5.f32[i] = 0;
      28              }
      29              else {
      30                 v5.u32[i] = v7.u32[i];
      31              }
      32          }
      33          else {
      34             v5.f32[i] = v1.f32[i] - v3.f32[i];
      35          }
      36  
      37          if (((1 << i) & m2) == 0) {
      38              if (zero_mask) {
      39                 v6.f32[i] = 0;
      40              }
      41              else {
      42                 v6.u32[i] = v8.u32[i];
      43              }
      44          }
      45          else {
      46              v6.f32[i] = v2.f32[i] - v4.f32[i];
      47          }
      48  
      49      }
      50      *dest = pack_twops_2ph(v5, v6);
      51  }
      52  
      53  
      54  void
      55  TEST (void)
      56  {
      57    V512 res;
      58    V512 exp;
      59  
      60    init_src();
      61    
      62    EMULATE(sub_ph) (&exp, src1, src2, NET_MASK, 0);
      63    HF(res) = INTRINSIC (_sub_ph) (HF(src1), HF(src2));
      64    CHECK_RESULT (&res, &exp, N_ELEMS, _sub_ph);
      65  
      66    init_dest(&res, &exp);
      67    EMULATE(sub_ph) (&exp, src1, src2, MASK_VALUE, 0);
      68    HF(res) = INTRINSIC (_mask_sub_ph) (HF(res), MASK_VALUE, HF(src1), HF(src2));
      69    CHECK_RESULT (&res, &exp, N_ELEMS, _mask_sub_ph);
      70  
      71    EMULATE(sub_ph) (&exp, src1, src2, ZMASK_VALUE, 1);
      72    HF(res) = INTRINSIC (_maskz_sub_ph) (ZMASK_VALUE, HF(src1), HF(src2));
      73    CHECK_RESULT (&res, &exp, N_ELEMS, _maskz_sub_ph);
      74  
      75  #if AVX512F_LEN == 512
      76    EMULATE(sub_ph) (&exp, src1, src2, NET_MASK, 0);
      77    HF(res) = INTRINSIC (_sub_round_ph) (HF(src1), HF(src2), _ROUND_NINT);
      78    CHECK_RESULT (&res, &exp, N_ELEMS, _sub_ph);
      79  
      80    init_dest(&res, &exp);
      81    EMULATE(sub_ph) (&exp, src1, src2, MASK_VALUE, 0);
      82    HF(res) = INTRINSIC (_mask_sub_round_ph) (HF(res), MASK_VALUE, HF(src1), HF(src2), _ROUND_NINT);
      83    CHECK_RESULT (&res, &exp, N_ELEMS, _mask_sub_ph);
      84  
      85    EMULATE(sub_ph) (&exp, src1, src2, ZMASK_VALUE, 1);
      86    HF(res) = INTRINSIC (_maskz_sub_round_ph) (ZMASK_VALUE, HF(src1), HF(src2), _ROUND_NINT);
      87    CHECK_RESULT (&res, &exp, N_ELEMS, _maskz_sub_ph);
      88  #endif
      89  
      90    if (n_errs != 0) {
      91        abort ();
      92    }
      93  }