|
|
dc03fd |
From 9c108bb84d3a2447dac730c455df658be0a2c751 Mon Sep 17 00:00:00 2001
|
|
|
dc03fd |
From: Richard Sandiford <richard.sandiford@arm.com>
|
|
|
dc03fd |
Date: Tue, 17 Aug 2021 15:15:27 +0100
|
|
|
dc03fd |
Subject: [PATCH] aarch64: Add -mtune=neoverse-512tvb
|
|
|
dc03fd |
To: gcc-patches@gcc.gnu.org
|
|
|
dc03fd |
|
|
|
dc03fd |
This patch adds an option to tune for Neoverse cores that have
|
|
|
dc03fd |
a total vector bandwidth of 512 bits (4x128 for Advanced SIMD
|
|
|
dc03fd |
and a vector-length-dependent equivalent for SVE). This is intended
|
|
|
dc03fd |
to be a compromise between tuning aggressively for a single core like
|
|
|
dc03fd |
Neoverse V1 (which can be too narrow) and tuning for AArch64 cores
|
|
|
dc03fd |
in general (which can be too wide).
|
|
|
dc03fd |
|
|
|
dc03fd |
-mcpu=neoverse-512tvb is equivalent to -mcpu=neoverse-v1
|
|
|
dc03fd |
-mtune=neoverse-512tvb.
|
|
|
dc03fd |
|
|
|
dc03fd |
gcc/
|
|
|
dc03fd |
* doc/invoke.texi: Document -mtune=neoverse-512tvb and
|
|
|
dc03fd |
-mcpu=neoverse-512tvb.
|
|
|
dc03fd |
* config/aarch64/aarch64-cores.def (neoverse-512tvb): New entry.
|
|
|
dc03fd |
* config/aarch64/aarch64-tune.md: Regenerate.
|
|
|
dc03fd |
|
|
|
dc03fd |
(cherry picked from commit 048039c49b96875144f67e7789fdea54abf7710b)
|
|
|
dc03fd |
---
|
|
|
dc03fd |
gcc/config/aarch64/aarch64-cores.def | 1 +
|
|
|
dc03fd |
gcc/config/aarch64/aarch64-tune.md | 2 +-
|
|
|
dc03fd |
gcc/doc/invoke.texi | 25 ++++++++++++++++++++++---
|
|
|
dc03fd |
3 files changed, 24 insertions(+), 4 deletions(-)
|
|
|
dc03fd |
|
|
|
dc03fd |
diff --git a/gcc/config/aarch64/aarch64-cores.def b/gcc/config/aarch64/aarch64-cores.def
|
|
|
dc03fd |
index dfb839c01cc..f348d31e22e 100644
|
|
|
dc03fd |
--- a/gcc/config/aarch64/aarch64-cores.def
|
|
|
dc03fd |
+++ b/gcc/config/aarch64/aarch64-cores.def
|
|
|
dc03fd |
@@ -99,6 +99,7 @@ AARCH64_CORE("saphira", saphira, falkor, 8_3A, AARCH64_FL_FOR_ARCH8_3
|
|
|
dc03fd |
/* ARM ('A') cores. */
|
|
|
dc03fd |
AARCH64_CORE("zeus", zeus, cortexa57, 8_4A, AARCH64_FL_FOR_ARCH8_4 | AARCH64_FL_F16 | AARCH64_FL_RCPC | AARCH64_FL_SVE | AARCH64_FL_RNG, neoversev1, 0x41, 0xd40, -1)
|
|
|
dc03fd |
AARCH64_CORE("neoverse-v1", neoversev1, cortexa57, 8_4A, AARCH64_FL_FOR_ARCH8_4 | AARCH64_FL_F16 | AARCH64_FL_RCPC | AARCH64_FL_SVE | AARCH64_FL_RNG, neoversev1, 0x41, 0xd40, -1)
|
|
|
dc03fd |
+AARCH64_CORE("neoverse-512tvb", neoverse512tvb, cortexa57, 8_4A, AARCH64_FL_FOR_ARCH8_4 | AARCH64_FL_F16 | AARCH64_FL_RCPC | AARCH64_FL_SVE | AARCH64_FL_RNG, neoversev1, INVALID_IMP, INVALID_CORE, -1)
|
|
|
dc03fd |
|
|
|
dc03fd |
/* Armv8.5-A Architecture Processors. */
|
|
|
dc03fd |
AARCH64_CORE("neoverse-n2", neoversen2, cortexa57, 8_4A, AARCH64_FL_FOR_ARCH8_4 | AARCH64_FL_F16 | AARCH64_FL_SVE | AARCH64_FL_RNG, neoversen2, 0x41, 0xd49, -1)
|
|
|
dc03fd |
diff --git a/gcc/config/aarch64/aarch64-tune.md b/gcc/config/aarch64/aarch64-tune.md
|
|
|
dc03fd |
index 2d7c9aa4740..09b76480f0b 100644
|
|
|
dc03fd |
--- a/gcc/config/aarch64/aarch64-tune.md
|
|
|
dc03fd |
+++ b/gcc/config/aarch64/aarch64-tune.md
|
|
|
dc03fd |
@@ -1,5 +1,5 @@
|
|
|
dc03fd |
;; -*- buffer-read-only: t -*-
|
|
|
dc03fd |
;; Generated automatically by gentune.sh from aarch64-cores.def
|
|
|
dc03fd |
(define_attr "tune"
|
|
|
dc03fd |
- "cortexa35,cortexa53,cortexa57,cortexa72,cortexa73,thunderx,thunderxt88p1,thunderxt88,thunderxt81,thunderxt83,xgene1,falkor,qdf24xx,exynosm1,thunderx2t99p1,vulcan,thunderx2t99,cortexa55,cortexa75,cortexa76,ares,neoversen1,saphira,zeus,neoversev1,neoversen2,cortexa57cortexa53,cortexa72cortexa53,cortexa73cortexa35,cortexa73cortexa53,cortexa75cortexa55"
|
|
|
dc03fd |
+ "cortexa35,cortexa53,cortexa57,cortexa72,cortexa73,thunderx,thunderxt88p1,thunderxt88,thunderxt81,thunderxt83,xgene1,falkor,qdf24xx,exynosm1,thunderx2t99p1,vulcan,thunderx2t99,cortexa55,cortexa75,cortexa76,ares,neoversen1,saphira,zeus,neoversev1,neoverse512tvb,neoversen2,cortexa57cortexa53,cortexa72cortexa53,cortexa73cortexa35,cortexa73cortexa53,cortexa75cortexa55"
|
|
|
dc03fd |
(const (symbol_ref "((enum attr_tune) aarch64_tune)")))
|
|
|
dc03fd |
diff --git a/gcc/doc/invoke.texi b/gcc/doc/invoke.texi
|
|
|
dc03fd |
index 78ca7738df2..68fda03281a 100644
|
|
|
dc03fd |
--- a/gcc/doc/invoke.texi
|
|
|
dc03fd |
+++ b/gcc/doc/invoke.texi
|
|
|
dc03fd |
@@ -14772,9 +14772,9 @@ performance of the code. Permissible values for this option are:
|
|
|
dc03fd |
@samp{generic}, @samp{cortex-a35}, @samp{cortex-a53}, @samp{cortex-a55},
|
|
|
dc03fd |
@samp{cortex-a57}, @samp{cortex-a72}, @samp{cortex-a73}, @samp{cortex-a75},
|
|
|
dc03fd |
@samp{cortex-a76}, @samp{ares}, @samp{neoverse-n1}, @samp{neoverse-n2},
|
|
|
dc03fd |
-@samp{neoverse-v1}, @samp{zeus}, @samp{exynos-m1}, @samp{falkor},
|
|
|
dc03fd |
-@samp{qdf24xx}, @samp{saphira}, @samp{xgene1}, @samp{vulcan}, @samp{thunderx},
|
|
|
dc03fd |
-@samp{thunderxt88}, @samp{thunderxt88p1}, @samp{thunderxt81},
|
|
|
dc03fd |
+@samp{neoverse-v1}, @samp{zeus}, @samp{neoverse-512tvb}, @samp{exynos-m1},
|
|
|
dc03fd |
+@samp{falkor}, @samp{qdf24xx}, @samp{saphira}, @samp{xgene1}, @samp{vulcan},
|
|
|
dc03fd |
+@samp{thunderx}, @samp{thunderxt88}, @samp{thunderxt88p1}, @samp{thunderxt81},
|
|
|
dc03fd |
@samp{thunderxt83}, @samp{thunderx2t99}, @samp{cortex-a57.cortex-a53},
|
|
|
dc03fd |
@samp{cortex-a72.cortex-a53}, @samp{cortex-a73.cortex-a35},
|
|
|
dc03fd |
@samp{cortex-a73.cortex-a53}, @samp{cortex-a75.cortex-a55},
|
|
|
dc03fd |
@@ -14785,6 +14785,15 @@ The values @samp{cortex-a57.cortex-a53}, @samp{cortex-a72.cortex-a53},
|
|
|
dc03fd |
@samp{cortex-a75.cortex-a55} specify that GCC should tune for a
|
|
|
dc03fd |
big.LITTLE system.
|
|
|
dc03fd |
|
|
|
dc03fd |
+The value @samp{neoverse-512tvb} specifies that GCC should tune
|
|
|
dc03fd |
+for Neoverse cores that (a) implement SVE and (b) have a total vector
|
|
|
dc03fd |
+bandwidth of 512 bits per cycle. In other words, the option tells GCC to
|
|
|
dc03fd |
+tune for Neoverse cores that can execute 4 128-bit Advanced SIMD arithmetic
|
|
|
dc03fd |
+instructions a cycle and that can execute an equivalent number of SVE
|
|
|
dc03fd |
+arithmetic instructions per cycle (2 for 256-bit SVE, 4 for 128-bit SVE).
|
|
|
dc03fd |
+This is more general than tuning for a specific core like Neoverse V1
|
|
|
dc03fd |
+but is more specific than the default tuning described below.
|
|
|
dc03fd |
+
|
|
|
dc03fd |
Additionally on native AArch64 GNU/Linux systems the value
|
|
|
dc03fd |
@samp{native} tunes performance to the host system. This option has no effect
|
|
|
dc03fd |
if the compiler is unable to recognize the processor of the host system.
|
|
|
dc03fd |
@@ -14814,6 +14823,16 @@ by @option{-mtune}). Where this option is used in conjunction
|
|
|
dc03fd |
with @option{-march} or @option{-mtune}, those options take precedence
|
|
|
dc03fd |
over the appropriate part of this option.
|
|
|
dc03fd |
|
|
|
dc03fd |
+@option{-mcpu=neoverse-512tvb} is special in that it does not refer
|
|
|
dc03fd |
+to a specific core, but instead refers to all Neoverse cores that
|
|
|
dc03fd |
+(a) implement SVE and (b) have a total vector bandwidth of 512 bits
|
|
|
dc03fd |
+a cycle. Unless overridden by @option{-march},
|
|
|
dc03fd |
+@option{-mcpu=neoverse-512tvb} generates code that can run on a
|
|
|
dc03fd |
+Neoverse V1 core, since Neoverse V1 is the first Neoverse core with
|
|
|
dc03fd |
+these properties. Unless overridden by @option{-mtune},
|
|
|
dc03fd |
+@option{-mcpu=neoverse-512tvb} tunes code in the same way as for
|
|
|
dc03fd |
+@option{-mtune=neoverse-512tvb}.
|
|
|
dc03fd |
+
|
|
|
dc03fd |
@item -moverride=@var{string}
|
|
|
dc03fd |
@opindex moverride
|
|
|
dc03fd |
Override tuning decisions made by the back-end in response to a
|
|
|
dc03fd |
--
|
|
|
dc03fd |
2.25.1
|
|
|
dc03fd |
|