diff options
Diffstat (limited to 'src/gallium/auxiliary/gallivm')
-rw-r--r-- | src/gallium/auxiliary/gallivm/lp_bld_init.c | 10 |
1 files changed, 9 insertions, 1 deletions
diff --git a/src/gallium/auxiliary/gallivm/lp_bld_init.c b/src/gallium/auxiliary/gallivm/lp_bld_init.c index 068a2cd7915..ffbe3eaed2c 100644 --- a/src/gallium/auxiliary/gallivm/lp_bld_init.c +++ b/src/gallium/auxiliary/gallivm/lp_bld_init.c @@ -434,8 +434,16 @@ lp_build_init(void) util_cpu_detect(); + /* AMD Bulldozer AVX's throughput is the same as SSE2; and because using + * 8-wide vector needs more floating ops than 4-wide (due to padding), it is + * actually more efficient to use 4-wide vectors on this processor. + * + * See also: + * - http://www.anandtech.com/show/4955/the-bulldozer-review-amd-fx8150-tested/2 + */ if (HAVE_AVX && - util_cpu_caps.has_avx) { + util_cpu_caps.has_avx && + util_cpu_caps.has_intel) { lp_native_vector_width = 256; } else { /* Leave it at 128, even when no SIMD extensions are available. |