lib/Headers/f16cintrin.h

*7330f729Sjoerg/*===---- f16cintrin.h - F16C intrinsics -----------------------------------===
*7330f729Sjoerg *
*7330f729Sjoerg * Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
*7330f729Sjoerg * See https://llvm.org/LICENSE.txt for license information.
*7330f729Sjoerg * SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
*7330f729Sjoerg *
*7330f729Sjoerg *===-----------------------------------------------------------------------===
*7330f729Sjoerg */
*7330f729Sjoerg
*7330f729Sjoerg#if !defined __IMMINTRIN_H
*7330f729Sjoerg#error "Never use <f16cintrin.h> directly; include <immintrin.h> instead."
*7330f729Sjoerg#endif
*7330f729Sjoerg
*7330f729Sjoerg#ifndef __F16CINTRIN_H
*7330f729Sjoerg#define __F16CINTRIN_H
*7330f729Sjoerg
*7330f729Sjoerg/* Define the default attributes for the functions in this file. */
*7330f729Sjoerg#define __DEFAULT_FN_ATTRS128 \
*7330f729Sjoerg  __attribute__((__always_inline__, __nodebug__, __target__("f16c"), __min_vector_width__(128)))
*7330f729Sjoerg#define __DEFAULT_FN_ATTRS256 \
*7330f729Sjoerg  __attribute__((__always_inline__, __nodebug__, __target__("f16c"), __min_vector_width__(256)))
*7330f729Sjoerg
*7330f729Sjoerg/* NOTE: Intel documents the 128-bit versions of these as being in emmintrin.h,
*7330f729Sjoerg * but that's because icc can emulate these without f16c using a library call.
*7330f729Sjoerg * Since we don't do that let's leave these in f16cintrin.h.
*7330f729Sjoerg */
*7330f729Sjoerg
*7330f729Sjoerg/// Converts a 16-bit half-precision float value into a 32-bit float
*7330f729Sjoerg///    value.
*7330f729Sjoerg///
*7330f729Sjoerg/// \headerfile <x86intrin.h>
*7330f729Sjoerg///
*7330f729Sjoerg/// This intrinsic corresponds to the <c> VCVTPH2PS </c> instruction.
*7330f729Sjoerg///
*7330f729Sjoerg/// \param __a
*7330f729Sjoerg///    A 16-bit half-precision float value.
*7330f729Sjoerg/// \returns The converted 32-bit float value.
*7330f729Sjoergstatic __inline float __DEFAULT_FN_ATTRS128
*7330f729Sjoerg_cvtsh_ss(unsigned short __a)
*7330f729Sjoerg{
*7330f729Sjoerg  __v8hi __v = {(short)__a, 0, 0, 0, 0, 0, 0, 0};
*7330f729Sjoerg  __v4sf __r = __builtin_ia32_vcvtph2ps(__v);
*7330f729Sjoerg  return __r[0];
*7330f729Sjoerg}
*7330f729Sjoerg
*7330f729Sjoerg/// Converts a 32-bit single-precision float value to a 16-bit
*7330f729Sjoerg///    half-precision float value.
*7330f729Sjoerg///
*7330f729Sjoerg/// \headerfile <x86intrin.h>
*7330f729Sjoerg///
*7330f729Sjoerg/// \code
*7330f729Sjoerg/// unsigned short _cvtss_sh(float a, const int imm);
*7330f729Sjoerg/// \endcode
*7330f729Sjoerg///
*7330f729Sjoerg/// This intrinsic corresponds to the <c> VCVTPS2PH </c> instruction.
*7330f729Sjoerg///
*7330f729Sjoerg/// \param a
*7330f729Sjoerg///    A 32-bit single-precision float value to be converted to a 16-bit
*7330f729Sjoerg///    half-precision float value.
*7330f729Sjoerg/// \param imm
*7330f729Sjoerg///    An immediate value controlling rounding using bits [2:0]: \n
*7330f729Sjoerg///    000: Nearest \n
*7330f729Sjoerg///    001: Down \n
*7330f729Sjoerg///    010: Up \n
*7330f729Sjoerg///    011: Truncate \n
*7330f729Sjoerg///    1XX: Use MXCSR.RC for rounding
*7330f729Sjoerg/// \returns The converted 16-bit half-precision float value.
*7330f729Sjoerg#define _cvtss_sh(a, imm) \
*7330f729Sjoerg  (unsigned short)(((__v8hi)__builtin_ia32_vcvtps2ph((__v4sf){a, 0, 0, 0}, \
*7330f729Sjoerg                                                     (imm)))[0])
*7330f729Sjoerg
*7330f729Sjoerg/// Converts a 128-bit vector containing 32-bit float values into a
*7330f729Sjoerg///    128-bit vector containing 16-bit half-precision float values.
*7330f729Sjoerg///
*7330f729Sjoerg/// \headerfile <x86intrin.h>
*7330f729Sjoerg///
*7330f729Sjoerg/// \code
*7330f729Sjoerg/// __m128i _mm_cvtps_ph(__m128 a, const int imm);
*7330f729Sjoerg/// \endcode
*7330f729Sjoerg///
*7330f729Sjoerg/// This intrinsic corresponds to the <c> VCVTPS2PH </c> instruction.
*7330f729Sjoerg///
*7330f729Sjoerg/// \param a
*7330f729Sjoerg///    A 128-bit vector containing 32-bit float values.
*7330f729Sjoerg/// \param imm
*7330f729Sjoerg///    An immediate value controlling rounding using bits [2:0]: \n
*7330f729Sjoerg///    000: Nearest \n
*7330f729Sjoerg///    001: Down \n
*7330f729Sjoerg///    010: Up \n
*7330f729Sjoerg///    011: Truncate \n
*7330f729Sjoerg///    1XX: Use MXCSR.RC for rounding
*7330f729Sjoerg/// \returns A 128-bit vector containing converted 16-bit half-precision float
*7330f729Sjoerg///    values. The lower 64 bits are used to store the converted 16-bit
*7330f729Sjoerg///    half-precision floating-point values.
*7330f729Sjoerg#define _mm_cvtps_ph(a, imm) \
*7330f729Sjoerg  (__m128i)__builtin_ia32_vcvtps2ph((__v4sf)(__m128)(a), (imm))
*7330f729Sjoerg
*7330f729Sjoerg/// Converts a 128-bit vector containing 16-bit half-precision float
*7330f729Sjoerg///    values into a 128-bit vector containing 32-bit float values.
*7330f729Sjoerg///
*7330f729Sjoerg/// \headerfile <x86intrin.h>
*7330f729Sjoerg///
*7330f729Sjoerg/// This intrinsic corresponds to the <c> VCVTPH2PS </c> instruction.
*7330f729Sjoerg///
*7330f729Sjoerg/// \param __a
*7330f729Sjoerg///    A 128-bit vector containing 16-bit half-precision float values. The lower
*7330f729Sjoerg///    64 bits are used in the conversion.
*7330f729Sjoerg/// \returns A 128-bit vector of [4 x float] containing converted float values.
*7330f729Sjoergstatic __inline __m128 __DEFAULT_FN_ATTRS128
*7330f729Sjoerg_mm_cvtph_ps(__m128i __a)
*7330f729Sjoerg{
*7330f729Sjoerg  return (__m128)__builtin_ia32_vcvtph2ps((__v8hi)__a);
*7330f729Sjoerg}
*7330f729Sjoerg
*7330f729Sjoerg/// Converts a 256-bit vector of [8 x float] into a 128-bit vector
*7330f729Sjoerg///    containing 16-bit half-precision float values.
*7330f729Sjoerg///
*7330f729Sjoerg/// \headerfile <x86intrin.h>
*7330f729Sjoerg///
*7330f729Sjoerg/// \code
*7330f729Sjoerg/// __m128i _mm256_cvtps_ph(__m256 a, const int imm);
*7330f729Sjoerg/// \endcode
*7330f729Sjoerg///
*7330f729Sjoerg/// This intrinsic corresponds to the <c> VCVTPS2PH </c> instruction.
*7330f729Sjoerg///
*7330f729Sjoerg/// \param a
*7330f729Sjoerg///    A 256-bit vector containing 32-bit single-precision float values to be
*7330f729Sjoerg///    converted to 16-bit half-precision float values.
*7330f729Sjoerg/// \param imm
*7330f729Sjoerg///    An immediate value controlling rounding using bits [2:0]: \n
*7330f729Sjoerg///    000: Nearest \n
*7330f729Sjoerg///    001: Down \n
*7330f729Sjoerg///    010: Up \n
*7330f729Sjoerg///    011: Truncate \n
*7330f729Sjoerg///    1XX: Use MXCSR.RC for rounding
*7330f729Sjoerg/// \returns A 128-bit vector containing the converted 16-bit half-precision
*7330f729Sjoerg///    float values.
*7330f729Sjoerg#define _mm256_cvtps_ph(a, imm) \
*7330f729Sjoerg (__m128i)__builtin_ia32_vcvtps2ph256((__v8sf)(__m256)(a), (imm))
*7330f729Sjoerg
*7330f729Sjoerg/// Converts a 128-bit vector containing 16-bit half-precision float
*7330f729Sjoerg///    values into a 256-bit vector of [8 x float].
*7330f729Sjoerg///
*7330f729Sjoerg/// \headerfile <x86intrin.h>
*7330f729Sjoerg///
*7330f729Sjoerg/// This intrinsic corresponds to the <c> VCVTPH2PS </c> instruction.
*7330f729Sjoerg///
*7330f729Sjoerg/// \param __a
*7330f729Sjoerg///    A 128-bit vector containing 16-bit half-precision float values to be
*7330f729Sjoerg///    converted to 32-bit single-precision float values.
*7330f729Sjoerg/// \returns A vector of [8 x float] containing the converted 32-bit
*7330f729Sjoerg///    single-precision float values.
*7330f729Sjoergstatic __inline __m256 __DEFAULT_FN_ATTRS256
*7330f729Sjoerg_mm256_cvtph_ps(__m128i __a)
*7330f729Sjoerg{
*7330f729Sjoerg  return (__m256)__builtin_ia32_vcvtph2ps256((__v8hi)__a);
*7330f729Sjoerg}
*7330f729Sjoerg
*7330f729Sjoerg#undef __DEFAULT_FN_ATTRS128
*7330f729Sjoerg#undef __DEFAULT_FN_ATTRS256
*7330f729Sjoerg
*7330f729Sjoerg#endif /* __F16CINTRIN_H */