Eigen  5.0.1
 
Loading...
Searching...
No Matches
Macros.h
1// This file is part of Eigen, a lightweight C++ template library
2// for linear algebra.
3//
4// Copyright (C) 2008-2015 Gael Guennebaud <gael.guennebaud@inria.fr>
5// Copyright (C) 2006-2008 Benoit Jacob <jacob.benoit.1@gmail.com>
6//
7// This Source Code Form is subject to the terms of the Mozilla
8// Public License v. 2.0. If a copy of the MPL was not distributed
9// with this file, You can obtain one at http://mozilla.org/MPL/2.0/.
10// SPDX-License-Identifier: MPL-2.0
11
12#ifndef EIGEN_MACROS_H
13#define EIGEN_MACROS_H
14// IWYU pragma: private
15#include "../InternalHeaderCheck.h"
16
17// for __cpp_lib feature test macros
18#if defined(__has_include) && __has_include(<version>)
19#include <version>
20#endif
21
22//------------------------------------------------------------------------------------------
23// Eigen version and basic defaults
24//------------------------------------------------------------------------------------------
25
26#define EIGEN_VERSION_AT_LEAST(x, y, z) \
27 (EIGEN_MAJOR_VERSION > x || \
28 (EIGEN_MAJOR_VERSION >= x && (EIGEN_MINOR_VERSION > y || (EIGEN_MINOR_VERSION >= y && EIGEN_PATCH_VERSION >= z))))
29
30#ifdef EIGEN_DEFAULT_TO_ROW_MAJOR
31#define EIGEN_DEFAULT_MATRIX_STORAGE_ORDER_OPTION Eigen::RowMajor
32#else
33#define EIGEN_DEFAULT_MATRIX_STORAGE_ORDER_OPTION Eigen::ColMajor
34#endif
35
36#ifndef EIGEN_DEFAULT_DENSE_INDEX_TYPE
37#define EIGEN_DEFAULT_DENSE_INDEX_TYPE std::ptrdiff_t
38#endif
39
40// Upperbound on the C++ version to use.
41// Expected values are 03, 11, 14, 17, etc.
42// By default, let's use an arbitrarily large C++ version.
43#ifndef EIGEN_MAX_CPP_VER
44#define EIGEN_MAX_CPP_VER 99
45#endif
46
52#ifndef EIGEN_FAST_MATH
53#define EIGEN_FAST_MATH 1
54#endif
55
56#ifndef EIGEN_STACK_ALLOCATION_LIMIT
57// 131072 == 128 KB
58#define EIGEN_STACK_ALLOCATION_LIMIT 131072
59// Marks the limit above as Eigen's own, so that a backend needing more room can raise it without
60// overriding a caller's stack-safety policy. ConfigureVectorization.h consumes and undefines it.
61#define EIGEN_STACK_ALLOCATION_LIMIT_WAS_DEFAULTED
62#endif
63
64//------------------------------------------------------------------------------------------
65// Compiler identification, EIGEN_COMP_*
66//------------------------------------------------------------------------------------------
67
69#ifdef __GNUC__
70#define EIGEN_COMP_GNUC (__GNUC__ * 100 + __GNUC_MINOR__ * 10 + __GNUC_PATCHLEVEL__)
71#else
72#define EIGEN_COMP_GNUC 0
73#endif
74
76#if defined(__clang__)
77#define EIGEN_COMP_CLANG (__clang_major__ * 100 + __clang_minor__ * 10 + __clang_patchlevel__)
78#else
79#define EIGEN_COMP_CLANG 0
80#endif
81
84#if defined(__clang__) && defined(__apple_build_version__)
85#define EIGEN_COMP_CLANGAPPLE __apple_build_version__
86#else
87#define EIGEN_COMP_CLANGAPPLE 0
88#endif
89
91#if defined(__castxml__)
92#define EIGEN_COMP_CASTXML 1
93#else
94#define EIGEN_COMP_CASTXML 0
95#endif
96
98#if defined(__llvm__)
99#define EIGEN_COMP_LLVM 1
100#else
101#define EIGEN_COMP_LLVM 0
102#endif
103
105#if defined(__INTEL_COMPILER)
106#define EIGEN_COMP_ICC __INTEL_COMPILER
107#else
108#define EIGEN_COMP_ICC 0
109#endif
110
112#if defined(__INTEL_CLANG_COMPILER)
113#define EIGEN_COMP_CLANGICC __INTEL_CLANG_COMPILER
114#else
115#define EIGEN_COMP_CLANGICC 0
116#endif
117
119#if defined(__MINGW32__)
120#define EIGEN_COMP_MINGW 1
121#else
122#define EIGEN_COMP_MINGW 0
123#endif
124
126#if defined(__SUNPRO_CC)
127#define EIGEN_COMP_SUNCC 1
128#else
129#define EIGEN_COMP_SUNCC 0
130#endif
131
133#if defined(_MSC_VER)
134#define EIGEN_COMP_MSVC _MSC_VER
135#else
136#define EIGEN_COMP_MSVC 0
137#endif
138
139#if defined(__NVCC__)
140// CUDA 11.4+ always defines __CUDACC_VER_MAJOR__.
141#define EIGEN_COMP_NVCC ((__CUDACC_VER_MAJOR__ * 10000) + (__CUDACC_VER_MINOR__ * 100))
142#else
143#define EIGEN_COMP_NVCC 0
144#endif
145
146// For the record, here is a table summarizing the possible values for EIGEN_COMP_MSVC:
147// name ver MSC_VER
148// 2015 14 1900
149// "15" 15 1900
150// 2017-14.1 15.0 1910
151// 2017-14.11 15.3 1911
152// 2017-14.12 15.5 1912
153// 2017-14.13 15.6 1913
154// 2017-14.14 15.7 1914
155// 2017 15.8 1915
156// 2017 15.9 1916
157// 2019 RTW 16.0 1920
158
160#if defined(_MSVC_LANG)
161#define EIGEN_COMP_MSVC_LANG _MSVC_LANG
162#else
163#define EIGEN_COMP_MSVC_LANG 0
164#endif
165
166// For the record, here is a table summarizing the possible values for EIGEN_COMP_MSVC_LANG:
167// MSVC option Standard MSVC_LANG
168// /std:c++14 (default as of VS 2019) C++14 201402L
169// /std:c++17 C++17 201703L
170// /std:c++latest >C++17 >201703L
171
174#if EIGEN_COMP_MSVC && !(EIGEN_COMP_ICC || EIGEN_COMP_LLVM || EIGEN_COMP_CLANG)
175#define EIGEN_COMP_MSVC_STRICT _MSC_VER
176#else
177#define EIGEN_COMP_MSVC_STRICT 0
178#endif
179
181// XLC version
182// 3.1 0x0301
183// 4.5 0x0405
184// 5.0 0x0500
185// 12.1 0x0C01
186#if defined(__IBMCPP__) || defined(__xlc__) || defined(__ibmxl__)
187#define EIGEN_COMP_IBM __xlC__
188#else
189#define EIGEN_COMP_IBM 0
190#endif
191
193#if defined(__PGI)
194#define EIGEN_COMP_PGI (__PGIC__ * 100 + __PGIC_MINOR__)
195#else
196#define EIGEN_COMP_PGI 0
197#endif
198
200#if defined(__NVCOMPILER)
201#define EIGEN_COMP_NVHPC (__NVCOMPILER_MAJOR__ * 100 + __NVCOMPILER_MINOR__)
202#else
203#define EIGEN_COMP_NVHPC 0
204#endif
205
207#if defined(__CC_ARM) || defined(__ARMCC_VERSION)
208#define EIGEN_COMP_ARM 1
209#else
210#define EIGEN_COMP_ARM 0
211#endif
212
214#if defined(__EMSCRIPTEN__)
215#define EIGEN_COMP_EMSCRIPTEN 1
216#else
217#define EIGEN_COMP_EMSCRIPTEN 0
218#endif
219
223#if defined(__FUJITSU)
224#define EIGEN_COMP_FCC (__FCC_major__ * 100 + __FCC_minor__ * 10 + __FCC_patchlevel__)
225#else
226#define EIGEN_COMP_FCC 0
227#endif
228
232#if defined(__CLANG_FUJITSU)
233#define EIGEN_COMP_CLANGFCC (__FCC_major__ * 100 + __FCC_minor__ * 10 + __FCC_patchlevel__)
234#else
235#define EIGEN_COMP_CLANGFCC 0
236#endif
237
241#if defined(_CRAYC) && !defined(__clang__)
242#define EIGEN_COMP_CPE (_RELEASE_MAJOR * 100 + _RELEASE_MINOR * 10 + _RELEASE_PATCHLEVEL)
243#else
244#define EIGEN_COMP_CPE 0
245#endif
246
250#if defined(_CRAYC) && defined(__clang__)
251#define EIGEN_COMP_CLANGCPE (_RELEASE_MAJOR * 100 + _RELEASE_MINOR * 10 + _RELEASE_PATCHLEVEL)
252#else
253#define EIGEN_COMP_CLANGCPE 0
254#endif
255
257#if defined(__LCC__) && defined(__MCST__)
258#define EIGEN_COMP_LCC (__LCC__ * 100 + __LCC_MINOR__)
259#else
260#define EIGEN_COMP_LCC 0
261#endif
262
265#if EIGEN_COMP_GNUC && \
266 !(EIGEN_COMP_CLANG || EIGEN_COMP_ICC || EIGEN_COMP_CLANGICC || EIGEN_COMP_MINGW || EIGEN_COMP_PGI || \
267 EIGEN_COMP_NVHPC || EIGEN_COMP_IBM || EIGEN_COMP_ARM || EIGEN_COMP_EMSCRIPTEN || EIGEN_COMP_FCC || \
268 EIGEN_COMP_CLANGFCC || EIGEN_COMP_CPE || EIGEN_COMP_CLANGCPE || EIGEN_COMP_LCC)
269#define EIGEN_COMP_GNUC_STRICT 1
270#else
271#define EIGEN_COMP_GNUC_STRICT 0
272#endif
273
274// GCC, and compilers that pretend to be it, have different version schemes, so this only makes sense to use with the
275// real GCC.
276#if EIGEN_COMP_GNUC_STRICT
277#define EIGEN_GNUC_STRICT_AT_LEAST(x, y, z) \
278 ((__GNUC__ > x) || (__GNUC__ == x && __GNUC_MINOR__ > y) || \
279 (__GNUC__ == x && __GNUC_MINOR__ == y && __GNUC_PATCHLEVEL__ >= z))
280#define EIGEN_GNUC_STRICT_LESS_THAN(x, y, z) \
281 ((__GNUC__ < x) || (__GNUC__ == x && __GNUC_MINOR__ < y) || \
282 (__GNUC__ == x && __GNUC_MINOR__ == y && __GNUC_PATCHLEVEL__ < z))
283#else
284#define EIGEN_GNUC_STRICT_AT_LEAST(x, y, z) 0
285#define EIGEN_GNUC_STRICT_LESS_THAN(x, y, z) 0
286#endif
287
288// Work around GCC PR tree-optimization/92420, which miscompiles some packetized complex arithmetic under -ffast-math.
289// The bug was introduced by GCC r238039, fixed on the GCC 8 branch by
290// https://gcc.gnu.org/g:785eda9390473e42f0e0b7199c42032a0432de68 and on the GCC 9 branch by
291// https://gcc.gnu.org/g:2d8ea3a0a6095a56b7c59c50b1068d602cde934a.
292// See also GitLab issues #1839 and #1840.
293#if defined(__FAST_MATH__) && EIGEN_COMP_GNUC_STRICT && EIGEN_GNUC_STRICT_AT_LEAST(7, 0, 0) && \
294 (EIGEN_GNUC_STRICT_LESS_THAN(8, 4, 0) || \
295 (EIGEN_GNUC_STRICT_AT_LEAST(9, 0, 0) && EIGEN_GNUC_STRICT_LESS_THAN(9, 3, 0)))
296#define EIGEN_GCC_FAST_MATH_COMPLEX_VECTORIZE_BUG 1
297// Disable the affected loop vectorizer around the complex packet helpers.
298#define EIGEN_GCC_FAST_MATH_COMPLEX_VECTORIZE_WORKAROUND_PUSH \
299 _Pragma("GCC push_options") _Pragma("GCC optimize(\"no-tree-loop-vectorize\")")
300#define EIGEN_GCC_FAST_MATH_COMPLEX_VECTORIZE_WORKAROUND_POP _Pragma("GCC pop_options")
301#else
302#define EIGEN_GCC_FAST_MATH_COMPLEX_VECTORIZE_BUG 0
303#define EIGEN_GCC_FAST_MATH_COMPLEX_VECTORIZE_WORKAROUND_PUSH
304#define EIGEN_GCC_FAST_MATH_COMPLEX_VECTORIZE_WORKAROUND_POP
305#endif
306
309#if EIGEN_COMP_CLANG && !(EIGEN_COMP_CLANGAPPLE || EIGEN_COMP_CLANGICC || EIGEN_COMP_CLANGFCC || EIGEN_COMP_CLANGCPE)
310#define EIGEN_COMP_CLANG_STRICT 1
311#else
312#define EIGEN_COMP_CLANG_STRICT 0
313#endif
314
315// Clang, and compilers forked from it, have different version schemes, so this only makes sense to use with the real
316// Clang.
317#if EIGEN_COMP_CLANG_STRICT
318#define EIGEN_CLANG_STRICT_AT_LEAST(x, y, z) \
319 ((__clang_major__ > x) || (__clang_major__ == x && __clang_minor__ > y) || \
320 (__clang_major__ == x && __clang_minor__ == y && __clang_patchlevel__ >= z))
321#define EIGEN_CLANG_STRICT_LESS_THAN(x, y, z) \
322 ((__clang_major__ < x) || (__clang_major__ == x && __clang_minor__ < y) || \
323 (__clang_major__ == x && __clang_minor__ == y && __clang_patchlevel__ < z))
324#else
325#define EIGEN_CLANG_STRICT_AT_LEAST(x, y, z) 0
326#define EIGEN_CLANG_STRICT_LESS_THAN(x, y, z) 0
327#endif
328
329//------------------------------------------------------------------------------------------
330// Architecture identification, EIGEN_ARCH_*
331//------------------------------------------------------------------------------------------
332
333#if defined(__x86_64__) || (defined(_M_X64) && !defined(_M_ARM64EC)) || defined(__amd64)
334#define EIGEN_ARCH_x86_64 1
335#else
336#define EIGEN_ARCH_x86_64 0
337#endif
338
339#if defined(__i386__) || defined(_M_IX86) || defined(_X86_) || defined(__i386)
340#define EIGEN_ARCH_i386 1
341#else
342#define EIGEN_ARCH_i386 0
343#endif
344
345#if EIGEN_ARCH_x86_64 || EIGEN_ARCH_i386
346#define EIGEN_ARCH_i386_OR_x86_64 1
347#else
348#define EIGEN_ARCH_i386_OR_x86_64 0
349#endif
350
352#if defined(__arm__)
353#define EIGEN_ARCH_ARM 1
354#else
355#define EIGEN_ARCH_ARM 0
356#endif
357
359#if defined(__aarch64__) || defined(_M_ARM64) || defined(_M_ARM64EC)
360#define EIGEN_ARCH_ARM64 1
361#else
362#define EIGEN_ARCH_ARM64 0
363#endif
364
366#if EIGEN_ARCH_ARM || EIGEN_ARCH_ARM64
367#define EIGEN_ARCH_ARM_OR_ARM64 1
368#else
369#define EIGEN_ARCH_ARM_OR_ARM64 0
370#endif
371
373#if EIGEN_ARCH_ARM_OR_ARM64 && defined(__ARM_ARCH) && __ARM_ARCH >= 8
374#define EIGEN_ARCH_ARMV8 1
375#else
376#define EIGEN_ARCH_ARMV8 0
377#endif
378
381#ifndef EIGEN_HAS_ARM64_FP16
382// NOTE: Older versions of Clang miscompile `__fp16` implicit conversions to `float`
383// on ARMv7 targets: <https://gitlab.com/libeigen/eigen/-/merge_requests/2273#note_3400347860>.
384#if EIGEN_ARCH_ARMV8 && defined(__ARM_FP16_FORMAT_IEEE)
385#define EIGEN_HAS_ARM64_FP16 1
386#else
387#define EIGEN_HAS_ARM64_FP16 0
388#endif
389#endif
390
392#if defined(__mips__) || defined(__mips)
393#define EIGEN_ARCH_MIPS 1
394#else
395#define EIGEN_ARCH_MIPS 0
396#endif
397
399#if defined(__loongarch64)
400#define EIGEN_ARCH_LOONGARCH64 1
401#else
402#define EIGEN_ARCH_LOONGARCH64 0
403#endif
404
406#if defined(__sparc__) || defined(__sparc)
407#define EIGEN_ARCH_SPARC 1
408#else
409#define EIGEN_ARCH_SPARC 0
410#endif
411
413#if defined(__ia64__)
414#define EIGEN_ARCH_IA64 1
415#else
416#define EIGEN_ARCH_IA64 0
417#endif
418
420#if defined(__powerpc__) || defined(__ppc__) || defined(_M_PPC) || defined(__POWERPC__)
421#define EIGEN_ARCH_PPC 1
422#else
423#define EIGEN_ARCH_PPC 0
424#endif
425
427#if defined(__riscv)
428#define EIGEN_ARCH_RISCV 1
429#else
430#define EIGEN_ARCH_RISCV 0
431#endif
432
433//------------------------------------------------------------------------------------------
434// Operating system identification, EIGEN_OS_*
435//------------------------------------------------------------------------------------------
436
438#if defined(__unix__) || defined(__unix)
439#define EIGEN_OS_UNIX 1
440#else
441#define EIGEN_OS_UNIX 0
442#endif
443
445#if defined(__linux__)
446#define EIGEN_OS_LINUX 1
447#else
448#define EIGEN_OS_LINUX 0
449#endif
450
452// note: ANDROID is defined when using ndk_build, __ANDROID__ is defined when using a standalone toolchain.
453#if defined(__ANDROID__) || defined(ANDROID)
454#define EIGEN_OS_ANDROID 1
455
456// Since NDK r16, `__NDK_MAJOR__` and `__NDK_MINOR__` are defined in
457// <android/ndk-version.h>. Include it when available so NDK-version-specific
458// workarounds can use these macros.
459#if defined __has_include
460#if __has_include(<android/ndk-version.h>)
461#include <android/ndk-version.h>
462#endif
463#endif
464
465#else
466#define EIGEN_OS_ANDROID 0
467#endif
468
470#if defined(__gnu_linux__) && !(EIGEN_OS_ANDROID)
471#define EIGEN_OS_GNULINUX 1
472#else
473#define EIGEN_OS_GNULINUX 0
474#endif
475
477#if defined(__FreeBSD__) || defined(__NetBSD__) || defined(__OpenBSD__) || defined(__bsdi__) || defined(__DragonFly__)
478#define EIGEN_OS_BSD 1
479#else
480#define EIGEN_OS_BSD 0
481#endif
482
484#if defined(__APPLE__)
485#define EIGEN_OS_MAC 1
486#else
487#define EIGEN_OS_MAC 0
488#endif
489
491#if defined(__QNX__)
492#define EIGEN_OS_QNX 1
493#else
494#define EIGEN_OS_QNX 0
495#endif
496
498#if defined(_WIN32)
499#define EIGEN_OS_WIN 1
500#else
501#define EIGEN_OS_WIN 0
502#endif
503
505#if defined(_WIN64)
506#define EIGEN_OS_WIN64 1
507#else
508#define EIGEN_OS_WIN64 0
509#endif
510
512#if defined(_WIN32_WCE)
513#define EIGEN_OS_WINCE 1
514#else
515#define EIGEN_OS_WINCE 0
516#endif
517
519#if defined(__CYGWIN__)
520#define EIGEN_OS_CYGWIN 1
521#else
522#define EIGEN_OS_CYGWIN 0
523#endif
524
526#if EIGEN_OS_WIN && !(EIGEN_OS_WINCE || EIGEN_OS_CYGWIN)
527#define EIGEN_OS_WIN_STRICT 1
528#else
529#define EIGEN_OS_WIN_STRICT 0
530#endif
531
533// compiler solaris __SUNPRO_C
534// version studio
535// 5.7 10 0x570
536// 5.8 11 0x580
537// 5.9 12 0x590
538// 5.10 12.1 0x5100
539// 5.11 12.2 0x5110
540// 5.12 12.3 0x5120
541#if (defined(sun) || defined(__sun)) && !(defined(__SVR4) || defined(__svr4__))
542#define EIGEN_OS_SUN __SUNPRO_C
543#else
544#define EIGEN_OS_SUN 0
545#endif
546
548#if (defined(sun) || defined(__sun)) && (defined(__SVR4) || defined(__svr4__))
549#define EIGEN_OS_SOLARIS 1
550#else
551#define EIGEN_OS_SOLARIS 0
552#endif
553
554//------------------------------------------------------------------------------------------
555// Detect GPU compilers and architectures
556//------------------------------------------------------------------------------------------
557
558// NVCC is not supported as the target platform for HIPCC
559// Note that this also makes EIGEN_CUDACC and EIGEN_HIPCC mutually exclusive
560#if defined(__NVCC__) && defined(__HIPCC__)
561#error "NVCC as the target platform for HIPCC is currently not supported."
562#endif
563
564#if defined(__CUDACC__) && !defined(EIGEN_NO_CUDA) && !defined(__SYCL_DEVICE_ONLY__)
565// Means the compiler is either nvcc or clang with CUDA enabled
566#define EIGEN_CUDACC __CUDACC__
567#endif
568
569#if defined(__CUDA_ARCH__) && !defined(EIGEN_NO_CUDA) && !defined(__SYCL_DEVICE_ONLY__)
570// Means we are generating code for the device
571#define EIGEN_CUDA_ARCH __CUDA_ARCH__
572#endif
573
574#if defined(EIGEN_CUDACC)
575#include <cuda.h>
576#define EIGEN_CUDA_SDK_VER (CUDA_VERSION * 10)
577#else
578#define EIGEN_CUDA_SDK_VER 0
579#endif
580
581// CUDA 11.8 is the oldest toolkit in GPU CI (it supports the sm_89 runners), not a new packet-intrinsic requirement.
582#if defined(EIGEN_CUDACC) && EIGEN_CUDA_SDK_VER > 0 && EIGEN_CUDA_SDK_VER < 110800
583#error "Eigen requires CUDA 11.8 or later."
584#endif
585
586// Native FP16 packet math intrinsics (e.g. __hfma2, h2exp, h2log) are only
587// declared by the CUDA headers when __CUDA_ARCH__ >= 530. Guard the device
588// pass with a clear error rather than surfacing as "identifier `__hfma2` is
589// undefined" deep inside PacketMath.h.
590#if defined(EIGEN_CUDA_ARCH) && EIGEN_CUDA_ARCH < 600
591#error "Eigen requires CUDA compute capability >= 6.0 (sm_60). Compile with -arch=sm_60 or higher."
592#endif
593
594#if defined(__HIPCC__) && !defined(EIGEN_NO_HIP) && !defined(__SYCL_DEVICE_ONLY__)
595// Means the compiler is HIPCC (analogous to EIGEN_CUDACC, but for HIP)
596#define EIGEN_HIPCC __HIPCC__
597
598// We need to include hip_runtime.h here because it pulls in
599// ++ hip_common.h which contains the define for __HIP_DEVICE_COMPILE__
600// ++ host_defines.h which contains the defines for the __host__ and __device__ macros
601#include <hip/hip_runtime.h>
602
603// Eigen requires ROCm/HIP >= 5.6 (GFX906 minimum architecture).
604// This floor exists to allow simplifying shared CUDA/HIP preprocessor guards —
605// all __HIP_ARCH_HAS_WARP_SHUFFLE__, __HIP_ARCH_HAS_FP16__, etc. are always true on GFX906+.
606#if defined(HIP_VERSION_MAJOR) && (HIP_VERSION_MAJOR < 5 || (HIP_VERSION_MAJOR == 5 && HIP_VERSION_MINOR < 6))
607#error "Eigen requires ROCm/HIP >= 5.6."
608#endif
609
610#if defined(__HIP_DEVICE_COMPILE__) && !defined(__SYCL_DEVICE_ONLY__)
611// analogous to EIGEN_CUDA_ARCH, but for HIP
612#define EIGEN_HIP_DEVICE_COMPILE __HIP_DEVICE_COMPILE__
613#endif
614
615#endif
616
617// Launch-bounds attribute under either GPU compiler, nothing elsewhere. Eigen's
618// kernels declare 1024 (EIGEN_HIP_LAUNCH_BOUNDS_1024 is the older spelling):
619// hipcc defaults to 256 and refuses larger blocks at launch, and nvcc without a
620// bound may allocate registers so that a 1024-thread launch fails with
621// cudaErrorLaunchOutOfResources (the 3D convolution kernel does).
622#if defined(EIGEN_CUDACC) || defined(EIGEN_HIPCC)
623#define EIGEN_GPU_LAUNCH_BOUNDS(n) __launch_bounds__(n)
624#else
625#define EIGEN_GPU_LAUNCH_BOUNDS(n)
626#endif
627#if !defined(EIGEN_HIP_LAUNCH_BOUNDS_1024)
628#define EIGEN_HIP_LAUNCH_BOUNDS_1024 EIGEN_GPU_LAUNCH_BOUNDS(1024)
629#endif
630
631// Unify CUDA/HIPCC
632
633#if defined(EIGEN_CUDACC) || defined(EIGEN_HIPCC)
634//
635// If either EIGEN_CUDACC or EIGEN_HIPCC is defined, then define EIGEN_GPUCC
636//
637#define EIGEN_GPUCC
638// NOTE: Some platforms (e.g. SPIRV) artificially set the CUDA SDK version to 0,
639// and don't support FP16, so we need to check the version number here. Every real toolkit is past the CUDA 11.8
640// floor checked above, so this only distinguishes "has a version" from "reports none".
641#if defined(EIGEN_CUDACC) && EIGEN_CUDA_SDK_VER > 0
642#define EIGEN_HAS_CUDA_FP16 1
643#elif defined(EIGEN_HIPCC)
644#define EIGEN_HAS_HIP_FP16 1
645#endif
646#if defined(EIGEN_HAS_CUDA_FP16) || defined(EIGEN_HAS_HIP_FP16)
647#define EIGEN_HAS_GPU_FP16 1
648#endif
649//
650// EIGEN_HIPCC implies the HIP compiler and is used to tweak Eigen code for use in HIP kernels
651// EIGEN_CUDACC implies the CUDA compiler and is used to tweak Eigen code for use in CUDA kernels
652//
653// In most cases the same tweaks are required to the Eigen code to enable in both the HIP and CUDA kernels.
654// For those cases, the corresponding code should be guarded with
655// #if defined(EIGEN_GPUCC)
656// instead of
657// #if defined(EIGEN_CUDACC) || defined(EIGEN_HIPCC)
658//
659// For cases where the tweak is specific to HIP, the code should be guarded with
660// #if defined(EIGEN_HIPCC)
661//
662// For cases where the tweak is specific to CUDA, the code should be guarded with
663// #if defined(EIGEN_CUDACC)
664//
665#endif
666
667#if defined(EIGEN_CUDA_ARCH) || defined(EIGEN_HIP_DEVICE_COMPILE)
668//
669// If either EIGEN_CUDA_ARCH or EIGEN_HIP_DEVICE_COMPILE is defined, then define EIGEN_GPU_COMPILE_PHASE
670//
671#define EIGEN_GPU_COMPILE_PHASE
672//
673// GPU compilers (HIPCC, NVCC) typically do two passes over the source code,
674// + one to compile the source for the "host" (ie CPU)
675// + another to compile the source for the "device" (ie. GPU)
676//
677// Code that needs to enabled only during the either the "host" or "device" compilation phase
678// needs to be guarded with a macro that indicates the current compilation phase
679//
680// EIGEN_HIP_DEVICE_COMPILE implies the device compilation phase in HIP
681// EIGEN_CUDA_ARCH implies the device compilation phase in CUDA
682//
683// In most cases, the "host" / "device" specific code is the same for both HIP and CUDA
684// For those cases, the code should be guarded with
685// #if defined(EIGEN_GPU_COMPILE_PHASE)
686// instead of
687// #if defined(EIGEN_CUDA_ARCH) || defined(EIGEN_HIP_DEVICE_COMPILE)
688//
689// For cases where the tweak is specific to HIP, the code should be guarded with
690// #if defined(EIGEN_HIP_DEVICE_COMPILE)
691//
692// For cases where the tweak is specific to CUDA, the code should be guarded with
693// #if defined(EIGEN_CUDA_ARCH)
694//
695#endif
696
699#if EIGEN_ARCH_ARM_OR_ARM64
700#ifndef EIGEN_HAS_ARM64_FP16_VECTOR_ARITHMETIC
701// Clang only supports FP16 on aarch64, and not all intrinsics are available
702// on A32 anyways even in GCC (e.g. vdiv_f16, vsqrt_f16).
703#if EIGEN_ARCH_ARM64 && defined(__ARM_FEATURE_FP16_VECTOR_ARITHMETIC) && !defined(EIGEN_GPU_COMPILE_PHASE)
704#define EIGEN_HAS_ARM64_FP16_VECTOR_ARITHMETIC 1
705#else
706#define EIGEN_HAS_ARM64_FP16_VECTOR_ARITHMETIC 0
707#endif
708#endif
709#endif
710
713#if EIGEN_ARCH_ARM_OR_ARM64
714#ifndef EIGEN_HAS_ARM64_FP16_SCALAR_ARITHMETIC
715// Clang only supports FP16 on aarch64, and not all intrinsics are available
716// on A32 anyways, even in GCC (e.g. vceqh_f16).
717#if EIGEN_ARCH_ARM64 && defined(__ARM_FEATURE_FP16_SCALAR_ARITHMETIC) && !defined(EIGEN_GPU_COMPILE_PHASE)
718#define EIGEN_HAS_ARM64_FP16_SCALAR_ARITHMETIC 1
719#endif
720#endif
721#endif
722
723#if defined(EIGEN_USE_SYCL) && defined(__SYCL_DEVICE_ONLY__)
724// EIGEN_USE_SYCL is a user-defined macro while __SYCL_DEVICE_ONLY__ is a compiler-defined macro.
725// In most cases we want to check if both macros are defined which can be done using the define below.
726#define SYCL_DEVICE_ONLY
727#endif
728
729// Under fast-math flags (-ffinite-math-only in particular), clang attaches the
730// `nofpclass(nan inf)` attribute to every function argument and return value of floating-point
731// (vector) type. A value that is a compile-time constant of NaN or infinity class -- e.g. the
732// all-ones bitmasks of ptrue and peven_mask, or the infinity/NaN constants guarding special
733// cases -- provably violates that attribute and is folded to poison, silently deleting the code
734// that consumes it. This barrier makes such a constant unprovable, at the cost of a stack
735// store/load pair where it is materialized. It uses a memory operand so that it works for any
736// packet type, including aggregates, and expands to nothing outside affected builds.
737#if EIGEN_COMP_CLANG && defined(__FINITE_MATH_ONLY__) && __FINITE_MATH_ONLY__ && !defined(EIGEN_GPU_COMPILE_PHASE) && \
738 !defined(SYCL_DEVICE_ONLY)
739#define EIGEN_FAST_MATH_CONSTANT_BARRIER(X) __asm__("" : "+m"(X))
740#else
741#define EIGEN_FAST_MATH_CONSTANT_BARRIER(X)
742#endif
743
744//------------------------------------------------------------------------------------------
745// Detect Compiler/Architecture/OS specific features
746//------------------------------------------------------------------------------------------
747
748// Cross compiler wrapper around LLVM's __has_builtin
749#ifdef __has_builtin
750#define EIGEN_HAS_BUILTIN(x) __has_builtin(x)
751#else
752#define EIGEN_HAS_BUILTIN(x) 0
753#endif
754
755// Cross compiler wrapper around LLVM's __has_attribute
756#ifdef __has_attribute
757#define EIGEN_HAS_ATTRIBUTE(x) __has_attribute(x)
758#else
759#define EIGEN_HAS_ATTRIBUTE(x) 0
760#endif
761
762// A Clang feature extension to determine compiler features.
763// We use it to determine 'cxx_rvalue_references'
764#ifndef __has_feature
765#define __has_feature(x) 0
766#endif
767
768// The macro EIGEN_CPLUSPLUS is a replacement for __cplusplus/_MSVC_LANG that
769// works for both platforms, indicating the C++ standard version number.
770//
771// With MSVC, without defining /Zc:__cplusplus, the __cplusplus macro will
772// report 199711L regardless of the language standard specified via /std.
773// We need to rely on _MSVC_LANG instead where available. Older MSVC versions
774// supported by Eigen do not define _MSVC_LANG, so use Eigen's minimum standard.
775#if EIGEN_COMP_MSVC_LANG > 0
776#define EIGEN_CPLUSPLUS EIGEN_COMP_MSVC_LANG
777#elif EIGEN_COMP_MSVC
778#define EIGEN_CPLUSPLUS 201402L
779#elif defined(__cplusplus)
780#define EIGEN_CPLUSPLUS __cplusplus
781#else
782#define EIGEN_CPLUSPLUS 0
783#endif
784
785// The macro EIGEN_COMP_CXXVER defines the c++ version expected by the compiler.
786// For instance, if compiling with gcc and -std=c++17, then EIGEN_COMP_CXXVER
787// is defined to 17.
788#if EIGEN_CPLUSPLUS >= 202002L
789#define EIGEN_COMP_CXXVER 20
790#elif EIGEN_CPLUSPLUS >= 201703L
791#define EIGEN_COMP_CXXVER 17
792#elif EIGEN_CPLUSPLUS >= 201402L
793#define EIGEN_COMP_CXXVER 14
794#else
795#define EIGEN_COMP_CXXVER 0
796#endif
797
798// The macros EIGEN_HAS_CXX?? defines a rough estimate of available c++ features
799// but in practice we should not rely on them but rather on the availability of
800// individual features as defined later.
801// This is why there is no EIGEN_HAS_CXX17.
802#if EIGEN_MAX_CPP_VER < 14 || EIGEN_COMP_CXXVER < 14 || (EIGEN_COMP_MSVC_STRICT && EIGEN_COMP_MSVC < 1910) || \
803 (EIGEN_COMP_ICC && EIGEN_COMP_ICC < 1700) || (EIGEN_COMP_CLANG_STRICT && EIGEN_COMP_CLANG < 390) || \
804 (EIGEN_COMP_CLANGAPPLE && EIGEN_COMP_CLANGAPPLE < 9000000) || (EIGEN_COMP_GNUC_STRICT && EIGEN_COMP_GNUC < 510)
805#error Eigen requires at least c++14 support.
806#endif
807
808// Deprecated compatibility macro. Eigen requires C++14 and no longer uses this
809// token internally, but keep it defined so downstream #if EIGEN_HAS_C99_MATH
810// checks do not trip -Wundef.
811#ifndef EIGEN_HAS_C99_MATH
812#define EIGEN_HAS_C99_MATH 1
813#endif
814
815// Does the compiler support std::hash?
816#ifndef EIGEN_HAS_STD_HASH
817// The std::hash struct is not labelled as a __device__ function and is not
818// constexpr, so cannot be used on device.
819#if !defined(EIGEN_GPU_COMPILE_PHASE)
820#define EIGEN_HAS_STD_HASH 1
821#else
822#define EIGEN_HAS_STD_HASH 0
823#endif
824#endif // EIGEN_HAS_STD_HASH
825
826#ifndef EIGEN_HAS_STD_INVOKE_RESULT
827#if EIGEN_MAX_CPP_VER >= 17 && EIGEN_COMP_CXXVER >= 17
828#define EIGEN_HAS_STD_INVOKE_RESULT 1
829#else
830#define EIGEN_HAS_STD_INVOKE_RESULT 0
831#endif
832#endif
833
834#define EIGEN_CONSTEXPR constexpr
835
836// NOTE: the required Apple's clang version is very conservative
837// and it could be that XCode 9 works just fine.
838// NOTE: the MSVC version is based on https://en.cppreference.com/w/cpp/compiler_support
839// and not tested.
840// NOTE: Intel C++ Compiler Classic (icc) Version 19.0 and later supports dynamic allocation
841// for over-aligned data, but not in a manner that is compatible with Eigen.
842// See https://gitlab.com/libeigen/eigen/-/issues/2575
843// Does the compiler support C++17 if constexpr?
844#ifndef EIGEN_HAS_CXX17_IFCONSTEXPR
845#if EIGEN_MAX_CPP_VER >= 17 && EIGEN_COMP_CXXVER >= 17 && \
846 ((EIGEN_COMP_MSVC >= 1911) || (EIGEN_GNUC_STRICT_AT_LEAST(7, 0, 0)) || EIGEN_COMP_CLANG_STRICT || \
847 (EIGEN_COMP_CLANGAPPLE && EIGEN_COMP_CLANGAPPLE >= 10000000))
848#define EIGEN_HAS_CXX17_IFCONSTEXPR 1
849#endif
850#endif
851
852#ifndef EIGEN_HAS_CXX17_OVERALIGN
853#if EIGEN_MAX_CPP_VER >= 17 && EIGEN_COMP_CXXVER >= 17 && \
854 ((EIGEN_COMP_MSVC >= 1912) || (EIGEN_GNUC_STRICT_AT_LEAST(7, 0, 0)) || (EIGEN_CLANG_STRICT_AT_LEAST(5, 0, 0)) || \
855 (EIGEN_COMP_CLANGAPPLE && EIGEN_COMP_CLANGAPPLE >= 10000000)) && \
856 !EIGEN_COMP_ICC
857#define EIGEN_HAS_CXX17_OVERALIGN 1
858#else
859#define EIGEN_HAS_CXX17_OVERALIGN 0
860#endif
861#endif
862
863#if defined(EIGEN_CUDACC)
864// Enable device-side constexpr when the toolchain supports relaxed constexpr rules.
865#if defined(__NVCC__)
866// nvcc considers constexpr functions as __host__ __device__ with the option --expt-relaxed-constexpr
867#ifdef __CUDACC_RELAXED_CONSTEXPR__
868#define EIGEN_CONSTEXPR_ARE_DEVICE_FUNC
869#endif
870#elif defined(__clang__) && defined(__CUDA__) && __has_feature(cxx_relaxed_constexpr)
871// clang++ always considers constexpr functions as implicitly __host__ __device__
872#define EIGEN_CONSTEXPR_ARE_DEVICE_FUNC
873#endif
874#endif
875
876// Does the compiler support the __int128 and __uint128_t extensions for 128-bit
877// integer arithmetic?
878//
879// Clang and GCC define __SIZEOF_INT128__ when these extensions are supported,
880// but we avoid using them in certain cases:
881//
882// * Building using Clang for Windows, where the Clang runtime library has
883// 128-bit support only on LP64 architectures, but Windows is LLP64.
884#ifndef EIGEN_HAS_BUILTIN_INT128
885#if defined(__SIZEOF_INT128__) && !(EIGEN_OS_WIN && EIGEN_COMP_CLANG)
886#define EIGEN_HAS_BUILTIN_INT128 1
887#else
888#define EIGEN_HAS_BUILTIN_INT128 0
889#endif
890#endif
891
892// Does the compiler support vector types?
893#if EIGEN_HAS_ATTRIBUTE(ext_vector_type) && EIGEN_HAS_BUILTIN(__builtin_vectorelements)
894#define EIGEN_ARCH_VECTOR_EXTENSIONS 1
895#else
896#define EIGEN_ARCH_VECTOR_EXTENSIONS 0
897#endif
898
899// Multidimensional subscript operator feature test
900#if defined(__cpp_multidimensional_subscript) && __cpp_multidimensional_subscript >= 202110L
901#define EIGEN_MULTIDIMENSIONAL_SUBSCRIPT
902#endif
903
904//------------------------------------------------------------------------------------------
905// Preprocessor programming helpers
906//------------------------------------------------------------------------------------------
907
908// This macro can be used to prevent from macro expansion, e.g.:
909// std::max EIGEN_NOT_A_MACRO(a,b)
910#define EIGEN_NOT_A_MACRO
911
912#define EIGEN_DEBUG_VAR(x) std::cerr << #x << " = " << x << std::endl;
913
914// concatenate two tokens
915#define EIGEN_CAT2(a, b) a##b
916#define EIGEN_CAT(a, b) EIGEN_CAT2(a, b)
917
918#define EIGEN_COMMA ,
919
920// convert a token to a string
921#define EIGEN_MAKESTRING2(a) #a
922#define EIGEN_MAKESTRING(a) EIGEN_MAKESTRING2(a)
923
924// EIGEN_STRONG_INLINE is a stronger version of the inline, using __forceinline on MSVC,
925// but it still doesn't use GCC's always_inline. This is useful in (common) situations where MSVC needs forceinline
926// but GCC is still doing fine with just inline.
927#ifndef EIGEN_STRONG_INLINE
928#if (EIGEN_COMP_MSVC || EIGEN_COMP_ICC) && !defined(EIGEN_GPUCC)
929#define EIGEN_STRONG_INLINE __forceinline
930#else
931#define EIGEN_STRONG_INLINE inline
932#endif
933#endif
934
935// EIGEN_ALWAYS_INLINE is the strongest default inline hint. It makes the function inline and, where supported,
936// adds attributes to maximize inlining. This should only be used when really necessary: in particular, the
937// __attribute__((always_inline)) used on GCC is often unnecessary and can severely harm compile times.
938#ifndef EIGEN_ALWAYS_INLINE
939#if EIGEN_COMP_GNUC && !defined(SYCL_DEVICE_ONLY)
940#define EIGEN_ALWAYS_INLINE __attribute__((always_inline)) inline
941#else
942#define EIGEN_ALWAYS_INLINE EIGEN_STRONG_INLINE
943#endif
944#endif
945
946// EIGEN_LAMBDA_ALWAYS_INLINE forces inlining of lambda functions.
947// On GCC/Clang, __attribute__((always_inline)) works on lambdas.
948// On MSVC, [[msvc::forceinline]] cannot be applied to generic lambdas
949// (those with auto parameters), so we leave it empty and rely on the
950// optimizer to inline small lambda bodies at /O2.
951#ifndef EIGEN_LAMBDA_ALWAYS_INLINE
952#if EIGEN_COMP_GNUC && !defined(SYCL_DEVICE_ONLY)
953#define EIGEN_LAMBDA_ALWAYS_INLINE __attribute__((always_inline))
954#else
955#define EIGEN_LAMBDA_ALWAYS_INLINE
956#endif
957#endif
958
959#ifndef EIGEN_DONT_INLINE
960#if EIGEN_COMP_GNUC
961#define EIGEN_DONT_INLINE __attribute__((noinline))
962#elif EIGEN_COMP_MSVC
963#define EIGEN_DONT_INLINE __declspec(noinline)
964#else
965#define EIGEN_DONT_INLINE
966#endif
967#endif
968
969#if EIGEN_COMP_GNUC
970#define EIGEN_PERMISSIVE_EXPR __extension__
971#else
972#define EIGEN_PERMISSIVE_EXPR
973#endif
974
975// GPU stuff
976
977// Disable some features when compiling with GPU compilers (SYCL/HIPCC)
978#if defined(SYCL_DEVICE_ONLY) || defined(EIGEN_HIP_DEVICE_COMPILE)
979// Do not try asserts on device code
980#ifndef EIGEN_NO_DEBUG
981#define EIGEN_NO_DEBUG
982#endif
983
984#ifdef EIGEN_INTERNAL_DEBUGGING
985#undef EIGEN_INTERNAL_DEBUGGING
986#endif
987#endif
988
989// No exceptions on device.
990#if defined(SYCL_DEVICE_ONLY) || defined(EIGEN_GPU_COMPILE_PHASE)
991#ifdef EIGEN_EXCEPTIONS
992#undef EIGEN_EXCEPTIONS
993#endif
994#endif
995
996#if defined(SYCL_DEVICE_ONLY)
997#ifndef EIGEN_DONT_VECTORIZE
998#define EIGEN_DONT_VECTORIZE
999#endif
1000#define EIGEN_DEVICE_FUNC __attribute__((flatten)) __attribute__((always_inline))
1001// All functions callable from CUDA/HIP code must be qualified with __device__
1002#elif defined(EIGEN_GPUCC)
1003#define EIGEN_DEVICE_FUNC __host__ __device__
1004#else
1005#define EIGEN_DEVICE_FUNC
1006#endif
1007
1008// this macro allows to get rid of linking errors about multiply defined functions.
1009// - static is not very good because it prevents definitions from different object files to be merged.
1010// So static causes the resulting linked executable to be bloated with multiple copies of the same function.
1011// - inline is not perfect either as it unwantedly hints the compiler toward inlining the function.
1012#define EIGEN_DECLARE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_DEVICE_FUNC
1013#define EIGEN_DEFINE_FUNCTION_ALLOWING_MULTIPLE_DEFINITIONS EIGEN_DEVICE_FUNC inline
1014
1015#ifdef NDEBUG
1016#ifndef EIGEN_NO_DEBUG
1017#define EIGEN_NO_DEBUG
1018#endif
1019#endif
1020
1021// eigen_assert can be overridden
1022#ifndef eigen_assert
1023#define eigen_assert(x) eigen_plain_assert(x)
1024#endif
1025
1026#ifdef EIGEN_INTERNAL_DEBUGGING
1027#define eigen_internal_assert(x) eigen_assert(x)
1028#else
1029#define eigen_internal_assert(x) ((void)0)
1030#endif
1031
1032#if defined(EIGEN_NO_DEBUG) || (defined(EIGEN_GPU_COMPILE_PHASE) && defined(EIGEN_NO_DEBUG_GPU))
1033#define EIGEN_ONLY_USED_FOR_DEBUG(x) EIGEN_UNUSED_VARIABLE(x)
1034#else
1035#define EIGEN_ONLY_USED_FOR_DEBUG(x)
1036#endif
1037
1038#ifndef EIGEN_NO_DEPRECATED_WARNING
1039#if EIGEN_COMP_GNUC
1040#define EIGEN_DEPRECATED __attribute__((deprecated))
1041#elif EIGEN_COMP_MSVC
1042#define EIGEN_DEPRECATED __declspec(deprecated)
1043#else
1044#define EIGEN_DEPRECATED
1045#endif
1046#else
1047#define EIGEN_DEPRECATED
1048#endif
1049
1050#ifndef EIGEN_NO_DEPRECATED_WARNING
1051#if EIGEN_COMP_GNUC
1052#define EIGEN_DEPRECATED_WITH_REASON(message) __attribute__((deprecated(message)))
1053#elif EIGEN_COMP_MSVC
1054#define EIGEN_DEPRECATED_WITH_REASON(message) __declspec(deprecated(message))
1055#else
1056#define EIGEN_DEPRECATED_WITH_REASON(message)
1057#endif
1058#else
1059#define EIGEN_DEPRECATED_WITH_REASON(message)
1060#endif
1061
1062// Deprecated no-op macro. Was a workaround for GCC 4.3 empty struct issues, removed in Eigen 5.0.
1063// Defined here for backward compatibility with downstream code that still references it.
1064#define EIGEN_EMPTY_STRUCT_CTOR(X)
1065
1066#if EIGEN_COMP_GNUC
1067#define EIGEN_UNUSED __attribute__((unused))
1068#else
1069#define EIGEN_UNUSED
1070#endif
1071
1072#if EIGEN_COMP_GNUC
1073#define EIGEN_PRAGMA(tokens) _Pragma(#tokens)
1074#define EIGEN_DIAGNOSTICS(tokens) EIGEN_PRAGMA(GCC diagnostic tokens)
1075#define EIGEN_DIAGNOSTICS_OFF(msc, gcc) EIGEN_DIAGNOSTICS(gcc)
1076#elif EIGEN_COMP_MSVC
1077#define EIGEN_PRAGMA(tokens) __pragma(tokens)
1078#define EIGEN_DIAGNOSTICS(tokens) EIGEN_PRAGMA(warning(tokens))
1079#define EIGEN_DIAGNOSTICS_OFF(msc, gcc) EIGEN_DIAGNOSTICS(msc)
1080#else
1081#define EIGEN_PRAGMA(tokens)
1082#define EIGEN_DIAGNOSTICS(tokens)
1083#define EIGEN_DIAGNOSTICS_OFF(msc, gcc)
1084#endif
1085
1086#define EIGEN_DISABLE_DEPRECATED_WARNING EIGEN_DIAGNOSTICS_OFF(disable : 4996, ignored "-Wdeprecated-declarations")
1087
1088// Suppresses 'unused variable' warnings.
1089namespace Eigen {
1090namespace internal {
1091template <typename T>
1092EIGEN_DEVICE_FUNC constexpr void ignore_unused_variable(const T&) {}
1093} // namespace internal
1094} // namespace Eigen
1095#define EIGEN_UNUSED_VARIABLE(var) Eigen::internal::ignore_unused_variable(var)
1096
1097#if !defined(EIGEN_ASM_COMMENT)
1098#if EIGEN_COMP_GNUC && (EIGEN_ARCH_i386_OR_x86_64 || EIGEN_ARCH_ARM_OR_ARM64 || EIGEN_ARCH_RISCV)
1099#define EIGEN_ASM_COMMENT(X) __asm__("#" X)
1100#else
1101#define EIGEN_ASM_COMMENT(X)
1102#endif
1103#endif
1104
1105// Acts as a barrier preventing operations involving `X` from crossing. This
1106// occurs, for example, in the fast rounding trick where a magic constant is
1107// added then subtracted, which is otherwise compiled away with -ffast-math.
1108//
1109// See bug 1674
1110#if defined(EIGEN_GPU_COMPILE_PHASE)
1111#define EIGEN_OPTIMIZATION_BARRIER(X)
1112#endif
1113
1114#if !defined(EIGEN_OPTIMIZATION_BARRIER)
1115// Implement the barrier on GNUC compilers or clang-cl.
1116#if EIGEN_COMP_GNUC || (defined(__clang__) && defined(_MSC_VER))
1117// According to https://gcc.gnu.org/onlinedocs/gcc/Constraints.html:
1118// X: Any operand whatsoever.
1119// r: A register operand is allowed provided that it is in a general
1120// register.
1121// g: Any register, memory or immediate integer operand is allowed, except
1122// for registers that are not general registers.
1123// w: (AArch32/AArch64) Floating point register, Advanced SIMD vector
1124// register or SVE vector register.
1125// x: (SSE) Any SSE register.
1126// (AArch64) Like w, but restricted to registers 0 to 15 inclusive.
1127// v: (PowerPC) An Altivec vector register.
1128// wa:(PowerPC) A VSX register.
1129//
1130// "X" (uppercase) should work for all cases, though this seems to fail for
1131// some versions of GCC for arm/aarch64 with
1132// "error: inconsistent operand constraints in an 'asm'"
1133// Clang x86_64/arm/aarch64 seems to require "g" to support both scalars and
1134// vectors, otherwise
1135// "error: non-trivial scalar-to-vector conversion, possible invalid
1136// constraint for vector type"
1137//
1138// GCC for ppc64le generates an internal compiler error with x/X/g.
1139// GCC for AVX generates an internal compiler error with X.
1140//
1141// Tested on icc/gcc/clang for sse, avx, avx2, avx512dq
1142// gcc for arm, aarch64,
1143// gcc for ppc64le,
1144// both vectors and scalars.
1145//
1146// Note that this is restricted to plain types - this will not work
1147// directly for std::complex<T>, Eigen::half, Eigen::bfloat16. For these,
1148// you will need to apply to the underlying POD type.
1149#if EIGEN_ARCH_PPC && EIGEN_COMP_GNUC_STRICT
1150// These register alternatives are broken on Clang. Packet4f is loaded into a single register rather than a vector,
1151// zeroing out some entries, and integer types generate a compile error.
1152#if EIGEN_OS_MAC
1153// General, Altivec for Apple (VSX were added in ISA v2.06):
1154#define EIGEN_PPC_OPTIMIZATION_BARRIER_REGISTERS "+r,v"
1155#else
1156// General, Altivec, VSX otherwise:
1157#define EIGEN_PPC_OPTIMIZATION_BARRIER_REGISTERS "+r,v,wa"
1158#endif
1159namespace Eigen {
1160namespace internal {
1161// Complex, class, and union operands go to memory: a class that must stay in memory is a compile error with the
1162// register constraints, and complex<float> comes back corrupted at -O3, even with "m" added as an alternative. The two
1163// constraints need separate instantiations because GCC rejects an impossible constraint even on a branch not taken.
1164template <bool InMemory>
1165struct ppc_optimization_barrier_impl {
1166 template <typename T>
1167 static EIGEN_ALWAYS_INLINE void run(T& x) {
1168 __asm__("" : EIGEN_PPC_OPTIMIZATION_BARRIER_REGISTERS(x));
1169 }
1170};
1171template <>
1172struct ppc_optimization_barrier_impl<true> {
1173 template <typename T>
1174 static EIGEN_ALWAYS_INLINE void run(T& x) {
1175 __asm__("" : "+m"(x));
1176 }
1177};
1178template <typename T>
1179EIGEN_ALWAYS_INLINE void ppc_optimization_barrier(T& x) {
1180 // __builtin_classify_type does not evaluate its operand: 9 = complex, 12 = class, 13 = union.
1181 constexpr int kTypeClass = __builtin_classify_type(*static_cast<T*>(nullptr));
1182 ppc_optimization_barrier_impl<kTypeClass == 9 || kTypeClass == 12 || kTypeClass == 13>::run(x);
1183}
1184} // namespace internal
1185} // namespace Eigen
1186#undef EIGEN_PPC_OPTIMIZATION_BARRIER_REGISTERS
1187#define EIGEN_OPTIMIZATION_BARRIER(X) Eigen::internal::ppc_optimization_barrier(X);
1188#elif EIGEN_ARCH_PPC && EIGEN_COMP_CLANG
1189// Clang's PPC backend does not accept one register constraint covering all scalar and vector operands. In particular,
1190// "wa" crashes the backend for scalar integers narrower than 64 bits.
1191#define EIGEN_OPTIMIZATION_BARRIER(X) __asm__("" : "+m"(X));
1192#elif EIGEN_ARCH_ARM_OR_ARM64
1193#ifdef __ARM_FP
1194// General, VFP or NEON.
1195// Clang doesn't like "r",
1196// error: non-trivial scalar-to-vector conversion, possible invalid
1197// constraint for vector typ
1198#define EIGEN_OPTIMIZATION_BARRIER(X) __asm__("" : "+g,w"(X));
1199#else
1200// Arm without VFP or NEON.
1201// "w" constraint will not compile.
1202#define EIGEN_OPTIMIZATION_BARRIER(X) __asm__("" : "+g"(X));
1203#endif
1204#elif EIGEN_ARCH_i386_OR_x86_64 && EIGEN_COMP_NVHPC
1205// nvc++ rejects a long double asm operand under every constraint, so its bytes go through memory instead.
1206namespace Eigen {
1207namespace internal {
1208template <typename T>
1209EIGEN_ALWAYS_INLINE void nvhpc_optimization_barrier(T& x) {
1210 __asm__("" : "+g,x"(x));
1211}
1212EIGEN_ALWAYS_INLINE void nvhpc_optimization_barrier(long double& x) {
1213 __asm__("" : "+m"(*reinterpret_cast<unsigned char(*)[sizeof(long double)]>(&x)));
1214}
1215} // namespace internal
1216} // namespace Eigen
1217#define EIGEN_OPTIMIZATION_BARRIER(X) Eigen::internal::nvhpc_optimization_barrier(X);
1218#elif EIGEN_ARCH_i386_OR_x86_64
1219// General, SSE.
1220#define EIGEN_OPTIMIZATION_BARRIER(X) __asm__("" : "+g,x"(X));
1221#elif !defined(SYCL_DEVICE_ONLY)
1222// Other architectures: a memory operand needs no target-specific register class.
1223#define EIGEN_OPTIMIZATION_BARRIER(X) __asm__("" : "+m"(X));
1224#else
1225#define EIGEN_OPTIMIZATION_BARRIER(X)
1226#endif
1227#else
1228// Not implemented for other compilers.
1229#define EIGEN_OPTIMIZATION_BARRIER(X)
1230#endif
1231#endif
1232
1233#if EIGEN_COMP_MSVC
1234// NOTE MSVC often gives C4127 warnings with compiletime if statements. See bug 1362.
1235// This workaround suppresses MSVC C4127 warnings for compile-time conditionals.
1236#define EIGEN_CONST_CONDITIONAL(cond) (void)0, cond
1237#else
1238#define EIGEN_CONST_CONDITIONAL(cond) cond
1239#endif
1240
1241#ifdef EIGEN_DONT_USE_RESTRICT_KEYWORD
1242#define EIGEN_RESTRICT
1243#endif
1244#ifndef EIGEN_RESTRICT
1245#define EIGEN_RESTRICT __restrict
1246#endif
1247
1248#ifndef EIGEN_DEFAULT_IO_FORMAT
1249#ifdef EIGEN_MAKING_DOCS
1250// format used in Eigen's documentation
1251// needed to define it here as escaping characters in CMake add_definition's argument seems very problematic.
1252#define EIGEN_DEFAULT_IO_FORMAT Eigen::IOFormat(3, 0, " ", "\n", "", "")
1253#else
1254#define EIGEN_DEFAULT_IO_FORMAT Eigen::IOFormat()
1255#endif
1256#endif
1257
1258// just an empty macro !
1259#define EIGEN_EMPTY
1260
1261// When compiling CUDA/HIP device code with NVCC or HIPCC
1262// pull in math functions from the global namespace.
1263// In host mode, and when device code is compiled with clang,
1264// use the std versions.
1265#if (defined(EIGEN_CUDA_ARCH) && defined(__NVCC__)) || defined(EIGEN_HIP_DEVICE_COMPILE)
1266#define EIGEN_USING_STD(FUNC) using ::FUNC;
1267#else
1268#define EIGEN_USING_STD(FUNC) using std::FUNC;
1269#endif
1270
1271#define EIGEN_INHERIT_ASSIGNMENT_EQUAL_OPERATOR(Derived) \
1272 using Base::operator=; \
1273 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Derived& operator=(const Derived& other) { \
1274 Base::operator=(other); \
1275 return *this; \
1276 }
1277
1283#define EIGEN_DEFAULT_COPY_CONSTRUCTOR(CLASS) EIGEN_DEVICE_FUNC CLASS(const CLASS&) = default;
1284
1290#define EIGEN_INHERIT_ASSIGNMENT_OPERATORS(Derived) \
1291 EIGEN_INHERIT_ASSIGNMENT_EQUAL_OPERATOR(Derived) \
1292 EIGEN_DEFAULT_COPY_CONSTRUCTOR(Derived)
1293
1299#define EIGEN_DEFAULT_EMPTY_CONSTRUCTOR_AND_DESTRUCTOR(Derived) \
1300 EIGEN_DEVICE_FUNC Derived() = default; \
1301 EIGEN_DEVICE_FUNC ~Derived() = default;
1302
1310
1311#define EIGEN_GENERIC_PUBLIC_INTERFACE(Derived) \
1312 typedef typename Eigen::internal::traits<Derived>::Scalar \
1313 Scalar; \
1314 typedef typename Eigen::NumTraits<Scalar>::Real \
1315 RealScalar; \
1317 typedef typename Base::CoeffReturnType \
1318 CoeffReturnType; \
1321 typedef typename Eigen::internal::ref_selector<Derived>::type Nested; \
1322 typedef typename Eigen::internal::traits<Derived>::StorageKind StorageKind; \
1323 typedef typename Eigen::internal::traits<Derived>::StorageIndex StorageIndex; \
1324 enum CompileTimeTraits { \
1325 RowsAtCompileTime = Eigen::internal::traits<Derived>::RowsAtCompileTime, \
1326 ColsAtCompileTime = Eigen::internal::traits<Derived>::ColsAtCompileTime, \
1327 Flags = Eigen::internal::traits<Derived>::Flags, \
1328 SizeAtCompileTime = Base::SizeAtCompileTime, \
1329 MaxSizeAtCompileTime = Base::MaxSizeAtCompileTime, \
1330 IsVectorAtCompileTime = Base::IsVectorAtCompileTime \
1331 }; \
1332 using Base::derived; \
1333 using Base::const_cast_derived;
1334
1335// FIXME Maybe the EIGEN_DENSE_PUBLIC_INTERFACE could be removed as importing PacketScalar is rarely needed
1336#define EIGEN_DENSE_PUBLIC_INTERFACE(Derived) \
1337 EIGEN_GENERIC_PUBLIC_INTERFACE(Derived) \
1338 typedef typename Base::PacketScalar PacketScalar;
1339
1340#if EIGEN_HAS_BUILTIN(__builtin_expect) || EIGEN_COMP_GNUC
1341#define EIGEN_PREDICT_FALSE(x) (__builtin_expect(x, false))
1342#define EIGEN_PREDICT_TRUE(x) (__builtin_expect(false || (x), true))
1343#else
1344#define EIGEN_PREDICT_FALSE(x) (x)
1345#define EIGEN_PREDICT_TRUE(x) (x)
1346#endif
1347
1348#define EIGEN_MAKE_CWISE_UNARY_OP(METHOD, FUNCTOR, RETURN_TYPE) \
1349 using RETURN_TYPE = CwiseUnaryOp<FUNCTOR<Scalar>, const Derived>; \
1350 EIGEN_DEVICE_FUNC constexpr EIGEN_STRONG_INLINE const RETURN_TYPE METHOD() const { return RETURN_TYPE(derived()); }
1351
1352// the expression type of a standard coefficient wise binary operation
1353#define EIGEN_CWISE_BINARY_RETURN_TYPE(LHS, RHS, FUNCTOR) \
1354 CwiseBinaryOp<FUNCTOR<typename internal::traits<LHS>::Scalar, typename internal::traits<RHS>::Scalar>, const LHS, \
1355 const RHS>
1356
1357#define EIGEN_MAKE_CWISE_BINARY_OP(METHOD, FUNCTOR) \
1358 template <typename OtherDerived> \
1359 EIGEN_DEVICE_FUNC constexpr EIGEN_STRONG_INLINE const EIGEN_CWISE_BINARY_RETURN_TYPE( \
1360 Derived, OtherDerived, FUNCTOR)(METHOD)(const EIGEN_CURRENT_STORAGE_BASE_CLASS<OtherDerived>& other) const { \
1361 return EIGEN_CWISE_BINARY_RETURN_TYPE(Derived, OtherDerived, FUNCTOR)(derived(), other.derived()); \
1362 }
1363
1364#define EIGEN_SCALAR_BINARY_SUPPORTED(FUNCTOR, TYPEA, TYPEB) \
1365 (Eigen::internal::has_ReturnType<Eigen::ScalarBinaryOpTraits<TYPEA, TYPEB, FUNCTOR<TYPEA, TYPEB> > >::value)
1366
1367#define EIGEN_EXPR_BINARYOP_SCALAR_RETURN_TYPE(EXPR, SCALAR, FUNCTOR) \
1368 CwiseBinaryOp<FUNCTOR<typename internal::traits<EXPR>::Scalar, SCALAR>, const EXPR, \
1369 const typename internal::plain_constant_type<EXPR, SCALAR>::type>
1370
1371#define EIGEN_SCALAR_BINARYOP_EXPR_RETURN_TYPE(SCALAR, EXPR, FUNCTOR) \
1372 CwiseBinaryOp<FUNCTOR<SCALAR, typename internal::traits<EXPR>::Scalar>, \
1373 const typename internal::plain_constant_type<EXPR, SCALAR>::type, const EXPR>
1374
1375#define EIGEN_MAKE_SCALAR_BINARY_OP_ONTHERIGHT(METHOD, FUNCTOR) \
1376 template <typename T> \
1377 EIGEN_DEVICE_FUNC constexpr EIGEN_STRONG_INLINE const EIGEN_EXPR_BINARYOP_SCALAR_RETURN_TYPE( \
1378 Derived, \
1379 typename internal::promote_scalar_arg<Scalar EIGEN_COMMA T EIGEN_COMMA EIGEN_SCALAR_BINARY_SUPPORTED( \
1380 FUNCTOR, Scalar, T)>::type, \
1381 FUNCTOR)(METHOD)(const T& scalar) const { \
1382 typedef typename internal::promote_scalar_arg<Scalar, T, EIGEN_SCALAR_BINARY_SUPPORTED(FUNCTOR, Scalar, T)>::type \
1383 PromotedT; \
1384 return EIGEN_EXPR_BINARYOP_SCALAR_RETURN_TYPE(Derived, PromotedT, FUNCTOR)( \
1385 derived(), typename internal::plain_constant_type<Derived, PromotedT>::type( \
1386 derived().rows(), derived().cols(), internal::scalar_constant_op<PromotedT>(scalar))); \
1387 }
1388
1389#define EIGEN_MAKE_SCALAR_BINARY_OP_ONTHELEFT(METHOD, FUNCTOR) \
1390 template <typename T> \
1391 EIGEN_DEVICE_FUNC constexpr EIGEN_STRONG_INLINE friend const EIGEN_SCALAR_BINARYOP_EXPR_RETURN_TYPE( \
1392 typename internal::promote_scalar_arg<Scalar EIGEN_COMMA T EIGEN_COMMA EIGEN_SCALAR_BINARY_SUPPORTED( \
1393 FUNCTOR, T, Scalar)>::type, \
1394 Derived, FUNCTOR)(METHOD)(const T& scalar, const StorageBaseType& matrix) { \
1395 typedef typename internal::promote_scalar_arg<Scalar, T, EIGEN_SCALAR_BINARY_SUPPORTED(FUNCTOR, T, Scalar)>::type \
1396 PromotedT; \
1397 return EIGEN_SCALAR_BINARYOP_EXPR_RETURN_TYPE(PromotedT, Derived, FUNCTOR)( \
1398 typename internal::plain_constant_type<Derived, PromotedT>::type( \
1399 matrix.derived().rows(), matrix.derived().cols(), internal::scalar_constant_op<PromotedT>(scalar)), \
1400 matrix.derived()); \
1401 }
1402
1403#define EIGEN_MAKE_SCALAR_BINARY_OP(METHOD, FUNCTOR) \
1404 EIGEN_MAKE_SCALAR_BINARY_OP_ONTHELEFT(METHOD, FUNCTOR) \
1405 EIGEN_MAKE_SCALAR_BINARY_OP_ONTHERIGHT(METHOD, FUNCTOR)
1406
1407#if (defined(_CPPUNWIND) || defined(__EXCEPTIONS)) && !defined(EIGEN_CUDA_ARCH) && !defined(EIGEN_EXCEPTIONS) && \
1408 !defined(EIGEN_USE_SYCL) && !defined(EIGEN_HIP_DEVICE_COMPILE)
1409#define EIGEN_EXCEPTIONS
1410#endif
1411
1412#ifdef EIGEN_EXCEPTIONS
1413#define EIGEN_THROW_X(X) throw X
1414#define EIGEN_THROW throw
1415#define EIGEN_TRY try
1416#define EIGEN_CATCH(X) catch (X)
1417#else
1418#if defined(EIGEN_CUDA_ARCH)
1419#define EIGEN_THROW_X(X) asm("trap;")
1420#define EIGEN_THROW asm("trap;")
1421#elif defined(EIGEN_HIP_DEVICE_COMPILE)
1422#define EIGEN_THROW_X(X) asm("s_trap 0")
1423#define EIGEN_THROW asm("s_trap 0")
1424#else
1425#define EIGEN_THROW_X(X) std::abort()
1426#define EIGEN_THROW std::abort()
1427#endif
1428#define EIGEN_TRY if (true)
1429#define EIGEN_CATCH(X) else
1430#endif
1431
1432// The all function is used to enable a variadic version of eigen_assert which can take a parameter pack as its input.
1433namespace Eigen {
1434namespace internal {
1435
1436EIGEN_DEVICE_FUNC constexpr bool all() { return true; }
1437
1438template <typename T, typename... Ts>
1439EIGEN_DEVICE_FUNC constexpr bool all(T t, Ts... ts) {
1440 return t && all(ts...);
1441}
1442
1443} // namespace internal
1444} // namespace Eigen
1445
1446// provide override and final specifiers if they are available:
1447#define EIGEN_OVERRIDE override
1448#define EIGEN_FINAL final
1449
1450// Wrapping #pragma unroll in a macro since it is required for SYCL
1451#if defined(SYCL_DEVICE_ONLY)
1452#if defined(_MSC_VER)
1453#define EIGEN_UNROLL_LOOP __pragma(unroll)
1454#else
1455#define EIGEN_UNROLL_LOOP _Pragma("unroll")
1456#endif
1457#else
1458#define EIGEN_UNROLL_LOOP
1459#endif
1460
1461// Notice: Use this macro with caution. The code in the if body should still
1462// compile with C++14.
1463#if defined(EIGEN_HAS_CXX17_IFCONSTEXPR)
1464#define EIGEN_IF_CONSTEXPR(...) if constexpr (__VA_ARGS__)
1465#else
1466#define EIGEN_IF_CONSTEXPR(...) if (__VA_ARGS__)
1467#endif
1468
1469#endif // EIGEN_MACROS_H