x86: Set default non_temporal_threshold for Zhaoxin processors

author MayShao-oc <MayShao-oc@zhaoxin.com>

Sat, 29 Jun 2024 03:58:28 +0000 (11:58 +0800)

committer H.J. Lu <hjl.tools@gmail.com>

Sun, 30 Jun 2024 13:26:43 +0000 (06:26 -0700)
author MayShao-oc <MayShao-oc@zhaoxin.com>
Sat, 29 Jun 2024 03:58:28 +0000 (11:58 +0800)
committer H.J. Lu <hjl.tools@gmail.com>
Sun, 30 Jun 2024 13:26:43 +0000 (06:26 -0700)
diff --git a/sysdeps/x86/cpu-features.c b/sysdeps/x86/cpu-features.c

index 1927f65699957f6762a44e4cfe8c5b64f771a103..e501e084ef40380db341dc1c9a4093d81af4719e 100644 (file)
--- a/sysdeps/x86/cpu-features.c
+++ b/sysdeps/x86/cpu-features.c
@@ -1065,6 +1065,7 @@ https://www.intel.com/content/www/us/en/support/articles/000059422/processors.ht
  
               /* Yongfeng and Shijidadao mircoarch tuning.  */
             case 0x5b:
+             cpu_features->cachesize_non_temporal_divisor = 2;
             case 0x6b:
               cpu_features->preferred[index_arch_AVX_Fast_Unaligned_Load]
                   &= ~bit_arch_AVX_Fast_Unaligned_Load;
diff --git a/sysdeps/x86/dl-cacheinfo.h b/sysdeps/x86/dl-cacheinfo.h

index 3a6ec4ef9f787e2fc439702985af9cafd1988640..5e77345a6e3998833ba7b2fbbb1c18ee78a902a1 100644 (file)
--- a/sysdeps/x86/dl-cacheinfo.h
+++ b/sysdeps/x86/dl-cacheinfo.h
@@ -934,8 +934,10 @@ dl_init_cacheinfo (struct cpu_features *cpu_features)
    /* If no ERMS, we use the per-thread L3 chunking. Normal cacheable stores run
       a higher risk of actually thrashing the cache as they don't have a HW LRU
       hint. As well, their performance in highly parallel situations is
-     noticeably worse.  */
-  if (!CPU_FEATURE_USABLE_P (cpu_features, ERMS))
+     noticeably worse. Zhaoxin processors are an exception, the lowbound is not
+     suitable for them based on actual test data.  */
+  if (!CPU_FEATURE_USABLE_P (cpu_features, ERMS)
+      && cpu_features->basic.kind != arch_kind_zhaoxin)
      non_temporal_threshold = non_temporal_threshold_lowbound;
    /* SIZE_MAX >> 4 because memmove-vec-unaligned-erms right-shifts the value of
       'x86_non_temporal_threshold' by `LOG_4X_MEMCPY_THRESH` (4) and it is best
author	MayShao-oc <MayShao-oc@zhaoxin.com>
	Sat, 29 Jun 2024 03:58:28 +0000 (11:58 +0800)
committer	H.J. Lu <hjl.tools@gmail.com>
	Sun, 30 Jun 2024 13:26:43 +0000 (06:26 -0700)
sysdeps/x86/cpu-features.c		patch \| blob \| blame \| history
sysdeps/x86/dl-cacheinfo.h		patch \| blob \| blame \| history