lib/Headers/f16cintrin.h

261991Sdim/*===---- f16cintrin.h - F16C intrinsics -----------------------------------===
243791Sdim *
353358Sdim * Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
353358Sdim * See https://llvm.org/LICENSE.txt for license information.
353358Sdim * SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
243791Sdim *
243791Sdim *===-----------------------------------------------------------------------===
243791Sdim */
243791Sdim
341825Sdim#if !defined __IMMINTRIN_H
341825Sdim#error "Never use <f16cintrin.h> directly; include <immintrin.h> instead."
243791Sdim#endif
243791Sdim
243791Sdim#ifndef __F16CINTRIN_H
243791Sdim#define __F16CINTRIN_H
243791Sdim
288943Sdim/* Define the default attributes for the functions in this file. */
341825Sdim#define __DEFAULT_FN_ATTRS128 \
341825Sdim  __attribute__((__always_inline__, __nodebug__, __target__("f16c"), __min_vector_width__(128)))
341825Sdim#define __DEFAULT_FN_ATTRS256 \
341825Sdim  __attribute__((__always_inline__, __nodebug__, __target__("f16c"), __min_vector_width__(256)))
288943Sdim
341825Sdim/* NOTE: Intel documents the 128-bit versions of these as being in emmintrin.h,
341825Sdim * but that's because icc can emulate these without f16c using a library call.
341825Sdim * Since we don't do that let's leave these in f16cintrin.h.
341825Sdim */
341825Sdim
341825Sdim/// Converts a 16-bit half-precision float value into a 32-bit float
309124Sdim///    value.
309124Sdim///
309124Sdim/// \headerfile <x86intrin.h>
309124Sdim///
314564Sdim/// This intrinsic corresponds to the <c> VCVTPH2PS </c> instruction.
309124Sdim///
309124Sdim/// \param __a
309124Sdim///    A 16-bit half-precision float value.
309124Sdim/// \returns The converted 32-bit float value.
341825Sdimstatic __inline float __DEFAULT_FN_ATTRS128
309124Sdim_cvtsh_ss(unsigned short __a)
309124Sdim{
353358Sdim  __v8hi __v = {(short)__a, 0, 0, 0, 0, 0, 0, 0};
353358Sdim  __v4sf __r = __builtin_ia32_vcvtph2ps(__v);
353358Sdim  return __r[0];
309124Sdim}
243791Sdim
341825Sdim/// Converts a 32-bit single-precision float value to a 16-bit
309124Sdim///    half-precision float value.
309124Sdim///
309124Sdim/// \headerfile <x86intrin.h>
309124Sdim///
309124Sdim/// \code
309124Sdim/// unsigned short _cvtss_sh(float a, const int imm);
309124Sdim/// \endcode
309124Sdim///
314564Sdim/// This intrinsic corresponds to the <c> VCVTPS2PH </c> instruction.
309124Sdim///
309124Sdim/// \param a
309124Sdim///    A 32-bit single-precision float value to be converted to a 16-bit
309124Sdim///    half-precision float value.
309124Sdim/// \param imm
314564Sdim///    An immediate value controlling rounding using bits [2:0]: \n
314564Sdim///    000: Nearest \n
314564Sdim///    001: Down \n
314564Sdim///    010: Up \n
314564Sdim///    011: Truncate \n
309124Sdim///    1XX: Use MXCSR.RC for rounding
309124Sdim/// \returns The converted 16-bit half-precision float value.
341825Sdim#define _cvtss_sh(a, imm) \
321369Sdim  (unsigned short)(((__v8hi)__builtin_ia32_vcvtps2ph((__v4sf){a, 0, 0, 0}, \
341825Sdim                                                     (imm)))[0])
309124Sdim
341825Sdim/// Converts a 128-bit vector containing 32-bit float values into a
309124Sdim///    128-bit vector containing 16-bit half-precision float values.
309124Sdim///
309124Sdim/// \headerfile <x86intrin.h>
309124Sdim///
309124Sdim/// \code
309124Sdim/// __m128i _mm_cvtps_ph(__m128 a, const int imm);
309124Sdim/// \endcode
309124Sdim///
314564Sdim/// This intrinsic corresponds to the <c> VCVTPS2PH </c> instruction.
309124Sdim///
309124Sdim/// \param a
309124Sdim///    A 128-bit vector containing 32-bit float values.
309124Sdim/// \param imm
314564Sdim///    An immediate value controlling rounding using bits [2:0]: \n
314564Sdim///    000: Nearest \n
314564Sdim///    001: Down \n
314564Sdim///    010: Up \n
314564Sdim///    011: Truncate \n
309124Sdim///    1XX: Use MXCSR.RC for rounding
309124Sdim/// \returns A 128-bit vector containing converted 16-bit half-precision float
309124Sdim///    values. The lower 64 bits are used to store the converted 16-bit
309124Sdim///    half-precision floating-point values.
341825Sdim#define _mm_cvtps_ph(a, imm) \
341825Sdim  (__m128i)__builtin_ia32_vcvtps2ph((__v4sf)(__m128)(a), (imm))
309124Sdim
341825Sdim/// Converts a 128-bit vector containing 16-bit half-precision float
309124Sdim///    values into a 128-bit vector containing 32-bit float values.
309124Sdim///
309124Sdim/// \headerfile <x86intrin.h>
309124Sdim///
314564Sdim/// This intrinsic corresponds to the <c> VCVTPH2PS </c> instruction.
309124Sdim///
309124Sdim/// \param __a
309124Sdim///    A 128-bit vector containing 16-bit half-precision float values. The lower
309124Sdim///    64 bits are used in the conversion.
309124Sdim/// \returns A 128-bit vector of [4 x float] containing converted float values.
341825Sdimstatic __inline __m128 __DEFAULT_FN_ATTRS128
249423Sdim_mm_cvtph_ps(__m128i __a)
243791Sdim{
249423Sdim  return (__m128)__builtin_ia32_vcvtph2ps((__v8hi)__a);
243791Sdim}
243791Sdim
341825Sdim/// Converts a 256-bit vector of [8 x float] into a 128-bit vector
341825Sdim///    containing 16-bit half-precision float values.
341825Sdim///
341825Sdim/// \headerfile <x86intrin.h>
341825Sdim///
341825Sdim/// \code
341825Sdim/// __m128i _mm256_cvtps_ph(__m256 a, const int imm);
341825Sdim/// \endcode
341825Sdim///
341825Sdim/// This intrinsic corresponds to the <c> VCVTPS2PH </c> instruction.
341825Sdim///
341825Sdim/// \param a
341825Sdim///    A 256-bit vector containing 32-bit single-precision float values to be
341825Sdim///    converted to 16-bit half-precision float values.
341825Sdim/// \param imm
341825Sdim///    An immediate value controlling rounding using bits [2:0]: \n
341825Sdim///    000: Nearest \n
341825Sdim///    001: Down \n
341825Sdim///    010: Up \n
341825Sdim///    011: Truncate \n
341825Sdim///    1XX: Use MXCSR.RC for rounding
341825Sdim/// \returns A 128-bit vector containing the converted 16-bit half-precision
341825Sdim///    float values.
341825Sdim#define _mm256_cvtps_ph(a, imm) \
341825Sdim (__m128i)__builtin_ia32_vcvtps2ph256((__v8sf)(__m256)(a), (imm))
288943Sdim
341825Sdim/// Converts a 128-bit vector containing 16-bit half-precision float
341825Sdim///    values into a 256-bit vector of [8 x float].
341825Sdim///
341825Sdim/// \headerfile <x86intrin.h>
341825Sdim///
341825Sdim/// This intrinsic corresponds to the <c> VCVTPH2PS </c> instruction.
341825Sdim///
341825Sdim/// \param __a
341825Sdim///    A 128-bit vector containing 16-bit half-precision float values to be
341825Sdim///    converted to 32-bit single-precision float values.
341825Sdim/// \returns A vector of [8 x float] containing the converted 32-bit
341825Sdim///    single-precision float values.
341825Sdimstatic __inline __m256 __DEFAULT_FN_ATTRS256
341825Sdim_mm256_cvtph_ps(__m128i __a)
341825Sdim{
341825Sdim  return (__m256)__builtin_ia32_vcvtph2ps256((__v8hi)__a);
341825Sdim}
341825Sdim
341825Sdim#undef __DEFAULT_FN_ATTRS128
341825Sdim#undef __DEFAULT_FN_ATTRS256
341825Sdim
243791Sdim#endif /* __F16CINTRIN_H */