[v1] aarch64: populate issue_info for the ampere1 family

Message ID 20260707212444.1552536-1-philipp.tomsich@vrull.eu
State New
Headers
Series [v1] aarch64: populate issue_info for the ampere1 family |

Checks

Context Check Description
linaro-tcwg-bot/tcwg_gcc_build--master-arm success Build passed
linaro-tcwg-bot/tcwg_simplebootstrap_build--master-arm-bootstrap fail Patch failed to apply
linaro-tcwg-bot/tcwg_gcc_build--master-aarch64 fail Patch failed to apply
linaro-tcwg-bot/tcwg_simplebootstrap_build--master-aarch64-bootstrap fail Patch failed to apply

Commit Message

Philipp Tomsich July 7, 2026, 9:24 p.m. UTC
  The ampere1, ampere1a and ampere1b tunes leave the vector issue_info
in cpu_vector_cost as NULL, so determine_suggested_unroll_factor bails
out early and reduction loops are never multi-accumulator-unrolled on
these cores.  Fill it in from the ampere1 optimization guide (L1D two
128-bit loads plus one store per cycle, four general ops per cycle).

gcc/ChangeLog:

	* config/aarch64/tuning_models/ampere1.h (ampere1_scalar_issue_info)
	(ampere1_advsimd_issue_info, ampere1_vec_issue_info): New.
	(ampere1_vector_cost): Use ampere1_vec_issue_info for issue_info.
	* config/aarch64/tuning_models/ampere1b.h (ampere1b_scalar_issue_info)
	(ampere1b_advsimd_issue_info, ampere1b_vec_issue_info): New.
	(ampere1b_vector_cost): Use ampere1b_vec_issue_info for issue_info.

gcc/testsuite/ChangeLog:

	* gcc.target/aarch64/ampere1-reduction-unroll-1.c: New test.

Signed-off-by: Philipp Tomsich <philipp.tomsich@vrull.eu>
---

 gcc/config/aarch64/tuning_models/ampere1.h    | 32 ++++++++++++++++++-
 gcc/config/aarch64/tuning_models/ampere1b.h   | 32 ++++++++++++++++++-
 .../aarch64/ampere1-reduction-unroll-1.c      | 20 ++++++++++++
 3 files changed, 82 insertions(+), 2 deletions(-)
 create mode 100644 gcc/testsuite/gcc.target/aarch64/ampere1-reduction-unroll-1.c
  

Comments

Andrea Pinski July 7, 2026, 9:37 p.m. UTC | #1
On Tue, Jul 7, 2026 at 2:25 PM Philipp Tomsich <philipp.tomsich@vrull.eu> wrote:
>
> The ampere1, ampere1a and ampere1b tunes leave the vector issue_info
> in cpu_vector_cost as NULL, so determine_suggested_unroll_factor bails
> out early and reduction loops are never multi-accumulator-unrolled on
> these cores.  Fill it in from the ampere1 optimization guide (L1D two
> 128-bit loads plus one store per cycle, four general ops per cycle).
>
> gcc/ChangeLog:
>
>         * config/aarch64/tuning_models/ampere1.h (ampere1_scalar_issue_info)
>         (ampere1_advsimd_issue_info, ampere1_vec_issue_info): New.
>         (ampere1_vector_cost): Use ampere1_vec_issue_info for issue_info.
>         * config/aarch64/tuning_models/ampere1b.h (ampere1b_scalar_issue_info)
>         (ampere1b_advsimd_issue_info, ampere1b_vec_issue_info): New.
>         (ampere1b_vector_cost): Use ampere1b_vec_issue_info for issue_info.
>
> gcc/testsuite/ChangeLog:
>
>         * gcc.target/aarch64/ampere1-reduction-unroll-1.c: New test.

LGTM.

>
> Signed-off-by: Philipp Tomsich <philipp.tomsich@vrull.eu>
> ---
>
>  gcc/config/aarch64/tuning_models/ampere1.h    | 32 ++++++++++++++++++-
>  gcc/config/aarch64/tuning_models/ampere1b.h   | 32 ++++++++++++++++++-
>  .../aarch64/ampere1-reduction-unroll-1.c      | 20 ++++++++++++
>  3 files changed, 82 insertions(+), 2 deletions(-)
>  create mode 100644 gcc/testsuite/gcc.target/aarch64/ampere1-reduction-unroll-1.c
>
> diff --git a/gcc/config/aarch64/tuning_models/ampere1.h b/gcc/config/aarch64/tuning_models/ampere1.h
> index 53d45761b42f..e2a42f585dd9 100644
> --- a/gcc/config/aarch64/tuning_models/ampere1.h
> +++ b/gcc/config/aarch64/tuning_models/ampere1.h
> @@ -46,6 +46,36 @@ static const advsimd_vec_cost ampere1_advsimd_vector_cost =
>    1  /* store_cost  */
>  };
>
> +static const aarch64_scalar_vec_issue_info ampere1_scalar_issue_info =
> +{
> +  3, /* loads_stores_per_cycle  */
> +  1, /* stores_per_cycle  */
> +  4, /* general_ops_per_cycle  */
> +  0, /* fp_simd_load_general_ops  */
> +  1 /* fp_simd_store_general_ops  */
> +};
> +
> +static const aarch64_advsimd_vec_issue_info ampere1_advsimd_issue_info =
> +{
> +  {
> +    2, /* loads_stores_per_cycle  */
> +    2, /* stores_per_cycle  */
> +    4, /* general_ops_per_cycle  */
> +    0, /* fp_simd_load_general_ops  */
> +    1 /* fp_simd_store_general_ops  */
> +  },
> +  2, /* ld2_st2_general_ops  */
> +  2, /* ld3_st3_general_ops  */
> +  3 /* ld4_st4_general_ops  */
> +};
> +
> +static const aarch64_vec_issue_info ampere1_vec_issue_info =
> +{
> +  &ampere1_scalar_issue_info,
> +  &ampere1_advsimd_issue_info,
> +  nullptr /* sve  */
> +};
> +
>  /* Ampere-1 costs for vector insn classes.  */
>  static const struct cpu_vector_cost ampere1_vector_cost =
>  {
> @@ -57,7 +87,7 @@ static const struct cpu_vector_cost ampere1_vector_cost =
>    1, /* cond_not_taken_branch_cost  */
>    &ampere1_advsimd_vector_cost, /* advsimd  */
>    nullptr, /* sve  */
> -  nullptr  /* issue_info  */
> +  &ampere1_vec_issue_info /* issue_info  */
>  };
>
>  static const cpu_prefetch_tune ampere1_prefetch_tune =
> diff --git a/gcc/config/aarch64/tuning_models/ampere1b.h b/gcc/config/aarch64/tuning_models/ampere1b.h
> index b84713b53668..8ba434d27bbf 100644
> --- a/gcc/config/aarch64/tuning_models/ampere1b.h
> +++ b/gcc/config/aarch64/tuning_models/ampere1b.h
> @@ -57,6 +57,36 @@ static const advsimd_vec_cost ampere1b_advsimd_vector_cost =
>    1  /* store_cost  */
>  };
>
> +static const aarch64_scalar_vec_issue_info ampere1b_scalar_issue_info =
> +{
> +  3, /* loads_stores_per_cycle  */
> +  1, /* stores_per_cycle  */
> +  4, /* general_ops_per_cycle  */
> +  0, /* fp_simd_load_general_ops  */
> +  1 /* fp_simd_store_general_ops  */
> +};
> +
> +static const aarch64_advsimd_vec_issue_info ampere1b_advsimd_issue_info =
> +{
> +  {
> +    2, /* loads_stores_per_cycle  */
> +    2, /* stores_per_cycle  */
> +    4, /* general_ops_per_cycle  */
> +    0, /* fp_simd_load_general_ops  */
> +    1 /* fp_simd_store_general_ops  */
> +  },
> +  2, /* ld2_st2_general_ops  */
> +  2, /* ld3_st3_general_ops  */
> +  3 /* ld4_st4_general_ops  */
> +};
> +
> +static const aarch64_vec_issue_info ampere1b_vec_issue_info =
> +{
> +  &ampere1b_scalar_issue_info,
> +  &ampere1b_advsimd_issue_info,
> +  nullptr /* sve  */
> +};
> +
>  /* Ampere-1B costs for vector insn classes.  */
>  static const struct cpu_vector_cost ampere1b_vector_cost =
>  {
> @@ -68,7 +98,7 @@ static const struct cpu_vector_cost ampere1b_vector_cost =
>    1, /* cond_not_taken_branch_cost  */
>    &ampere1b_advsimd_vector_cost, /* advsimd  */
>    nullptr, /* sve  */
> -  nullptr  /* issue_info  */
> +  &ampere1b_vec_issue_info /* issue_info  */
>  };
>
>  static const struct tune_params ampere1b_tunings =
> diff --git a/gcc/testsuite/gcc.target/aarch64/ampere1-reduction-unroll-1.c b/gcc/testsuite/gcc.target/aarch64/ampere1-reduction-unroll-1.c
> new file mode 100644
> index 000000000000..61ab57304894
> --- /dev/null
> +++ b/gcc/testsuite/gcc.target/aarch64/ampere1-reduction-unroll-1.c
> @@ -0,0 +1,20 @@
> +/* { dg-do compile } */
> +/* { dg-options "-Ofast -mcpu=ampere1" } */
> +
> +/* The ampere1 family used a nullptr for the vector issue-info struct, so
> +   aarch64_vector_costs::determine_suggested_unroll_factor returned 1 at the
> +   `if (!vec_issue) return 1;' guard and reduction loops were never
> +   multi-accumulator-unrolled.  With issue_info populated, a unit-stride FP
> +   dot product widens to a suggested unroll factor of 4 and emits four
> +   independent `fmla v.4s' accumulator chains in the inner loop.  */
> +
> +float
> +dot (const float *a, const float *b, int n)
> +{
> +  float s = 0.0f;
> +  for (int i = 0; i < n; i++)
> +    s += a[i] * b[i];
> +  return s;
> +}
> +
> +/* { dg-final { scan-assembler-times {\tfmla\tv[0-9]+\.4s} 4 } } */
> --
> 2.34.1
>
> base-commit: 3b443e764144526748d2b51d017d745f1956377e
  
Kyrylo Tkachov July 8, 2026, 6:19 a.m. UTC | #2
Ok but a comment nit inline...

> On 7 Jul 2026, at 23:37, Andrea Pinski <andrew.pinski@oss.qualcomm.com> wrote:
> 
> On Tue, Jul 7, 2026 at 2:25 PM Philipp Tomsich <philipp.tomsich@vrull.eu> wrote:
>> 
>> The ampere1, ampere1a and ampere1b tunes leave the vector issue_info
>> in cpu_vector_cost as NULL, so determine_suggested_unroll_factor bails
>> out early and reduction loops are never multi-accumulator-unrolled on
>> these cores.  Fill it in from the ampere1 optimization guide (L1D two
>> 128-bit loads plus one store per cycle, four general ops per cycle).
>> 
>> gcc/ChangeLog:
>> 
>>        * config/aarch64/tuning_models/ampere1.h (ampere1_scalar_issue_info)
>>        (ampere1_advsimd_issue_info, ampere1_vec_issue_info): New.
>>        (ampere1_vector_cost): Use ampere1_vec_issue_info for issue_info.
>>        * config/aarch64/tuning_models/ampere1b.h (ampere1b_scalar_issue_info)
>>        (ampere1b_advsimd_issue_info, ampere1b_vec_issue_info): New.
>>        (ampere1b_vector_cost): Use ampere1b_vec_issue_info for issue_info.
>> 
>> gcc/testsuite/ChangeLog:
>> 
>>        * gcc.target/aarch64/ampere1-reduction-unroll-1.c: New test.
> 
> LGTM.
> 
>> 
>> Signed-off-by: Philipp Tomsich <philipp.tomsich@vrull.eu>
>> ---
>> 
>> gcc/config/aarch64/tuning_models/ampere1.h    | 32 ++++++++++++++++++-
>> gcc/config/aarch64/tuning_models/ampere1b.h   | 32 ++++++++++++++++++-
>> .../aarch64/ampere1-reduction-unroll-1.c      | 20 ++++++++++++
>> 3 files changed, 82 insertions(+), 2 deletions(-)
>> create mode 100644 gcc/testsuite/gcc.target/aarch64/ampere1-reduction-unroll-1.c
>> 
>> diff --git a/gcc/config/aarch64/tuning_models/ampere1.h b/gcc/config/aarch64/tuning_models/ampere1.h
>> index 53d45761b42f..e2a42f585dd9 100644
>> --- a/gcc/config/aarch64/tuning_models/ampere1.h
>> +++ b/gcc/config/aarch64/tuning_models/ampere1.h
>> @@ -46,6 +46,36 @@ static const advsimd_vec_cost ampere1_advsimd_vector_cost =
>>   1  /* store_cost  */
>> };
>> 
>> +static const aarch64_scalar_vec_issue_info ampere1_scalar_issue_info =
>> +{
>> +  3, /* loads_stores_per_cycle  */
>> +  1, /* stores_per_cycle  */
>> +  4, /* general_ops_per_cycle  */
>> +  0, /* fp_simd_load_general_ops  */
>> +  1 /* fp_simd_store_general_ops  */
>> +};
>> +
>> +static const aarch64_advsimd_vec_issue_info ampere1_advsimd_issue_info =
>> +{
>> +  {
>> +    2, /* loads_stores_per_cycle  */
>> +    2, /* stores_per_cycle  */
>> +    4, /* general_ops_per_cycle  */
>> +    0, /* fp_simd_load_general_ops  */
>> +    1 /* fp_simd_store_general_ops  */
>> +  },
>> +  2, /* ld2_st2_general_ops  */
>> +  2, /* ld3_st3_general_ops  */
>> +  3 /* ld4_st4_general_ops  */
>> +};
>> +
>> +static const aarch64_vec_issue_info ampere1_vec_issue_info =
>> +{
>> +  &ampere1_scalar_issue_info,
>> +  &ampere1_advsimd_issue_info,
>> +  nullptr /* sve  */
>> +};
>> +
>> /* Ampere-1 costs for vector insn classes.  */
>> static const struct cpu_vector_cost ampere1_vector_cost =
>> {
>> @@ -57,7 +87,7 @@ static const struct cpu_vector_cost ampere1_vector_cost =
>>   1, /* cond_not_taken_branch_cost  */
>>   &ampere1_advsimd_vector_cost, /* advsimd  */
>>   nullptr, /* sve  */
>> -  nullptr  /* issue_info  */
>> +  &ampere1_vec_issue_info /* issue_info  */
>> };
>> 
>> static const cpu_prefetch_tune ampere1_prefetch_tune =
>> diff --git a/gcc/config/aarch64/tuning_models/ampere1b.h b/gcc/config/aarch64/tuning_models/ampere1b.h
>> index b84713b53668..8ba434d27bbf 100644
>> --- a/gcc/config/aarch64/tuning_models/ampere1b.h
>> +++ b/gcc/config/aarch64/tuning_models/ampere1b.h
>> @@ -57,6 +57,36 @@ static const advsimd_vec_cost ampere1b_advsimd_vector_cost =
>>   1  /* store_cost  */
>> };
>> 
>> +static const aarch64_scalar_vec_issue_info ampere1b_scalar_issue_info =
>> +{
>> +  3, /* loads_stores_per_cycle  */
>> +  1, /* stores_per_cycle  */
>> +  4, /* general_ops_per_cycle  */
>> +  0, /* fp_simd_load_general_ops  */
>> +  1 /* fp_simd_store_general_ops  */
>> +};
>> +
>> +static const aarch64_advsimd_vec_issue_info ampere1b_advsimd_issue_info =
>> +{
>> +  {
>> +    2, /* loads_stores_per_cycle  */
>> +    2, /* stores_per_cycle  */
>> +    4, /* general_ops_per_cycle  */
>> +    0, /* fp_simd_load_general_ops  */
>> +    1 /* fp_simd_store_general_ops  */
>> +  },
>> +  2, /* ld2_st2_general_ops  */
>> +  2, /* ld3_st3_general_ops  */
>> +  3 /* ld4_st4_general_ops  */
>> +};
>> +
>> +static const aarch64_vec_issue_info ampere1b_vec_issue_info =
>> +{
>> +  &ampere1b_scalar_issue_info,
>> +  &ampere1b_advsimd_issue_info,
>> +  nullptr /* sve  */
>> +};
>> +
>> /* Ampere-1B costs for vector insn classes.  */
>> static const struct cpu_vector_cost ampere1b_vector_cost =
>> {
>> @@ -68,7 +98,7 @@ static const struct cpu_vector_cost ampere1b_vector_cost =
>>   1, /* cond_not_taken_branch_cost  */
>>   &ampere1b_advsimd_vector_cost, /* advsimd  */
>>   nullptr, /* sve  */
>> -  nullptr  /* issue_info  */
>> +  &ampere1b_vec_issue_info /* issue_info  */
>> };
>> 
>> static const struct tune_params ampere1b_tunings =
>> diff --git a/gcc/testsuite/gcc.target/aarch64/ampere1-reduction-unroll-1.c b/gcc/testsuite/gcc.target/aarch64/ampere1-reduction-unroll-1.c
>> new file mode 100644
>> index 000000000000..61ab57304894
>> --- /dev/null
>> +++ b/gcc/testsuite/gcc.target/aarch64/ampere1-reduction-unroll-1.c
>> @@ -0,0 +1,20 @@
>> +/* { dg-do compile } */
>> +/* { dg-options "-Ofast -mcpu=ampere1" } */
>> +
>> +/* The ampere1 family used a nullptr for the vector issue-info struct, so
>> +   aarch64_vector_costs::determine_suggested_unroll_factor returned 1 at the
>> +   `if (!vec_issue) return 1;' guard and reduction loops were never
>> +   multi-accumulator-unrolled.  With issue_info populated, a unit-stride FP
>> +   dot product widens to a suggested unroll factor of 4 and emits four
>> +   independent `fmla v.4s' accumulator chains in the inner loop.  */

I don’t find it useful for comments to talk about what the code used to be, they should describe what the code does now. I suggest you greatly simplify this comment and just say that for ampere1 you expect this loop to have unroll factor 4, or something to that effect.
Thanks,
Kyrill


>> +
>> +float
>> +dot (const float *a, const float *b, int n)
>> +{
>> +  float s = 0.0f;
>> +  for (int i = 0; i < n; i++)
>> +    s += a[i] * b[i];
>> +  return s;
>> +}
>> +
>> +/* { dg-final { scan-assembler-times {\tfmla\tv[0-9]+\.4s} 4 } } */
>> --
>> 2.34.1
>> 
>> base-commit: 3b443e764144526748d2b51d017d745f1956377e
  
Philipp Tomsich July 8, 2026, 7:10 a.m. UTC | #3
Applied (with a shortened comment) to master as 7e35df8c412. Thanks!
-- Philipp.


On Wed, 8 Jul 2026 at 08:19, Kyrylo Tkachov <ktkachov@nvidia.com> wrote:
>
> Ok but a comment nit inline...
>
> > On 7 Jul 2026, at 23:37, Andrea Pinski <andrew.pinski@oss.qualcomm.com> wrote:
> >
> > On Tue, Jul 7, 2026 at 2:25 PM Philipp Tomsich <philipp.tomsich@vrull.eu> wrote:
> >>
> >> The ampere1, ampere1a and ampere1b tunes leave the vector issue_info
> >> in cpu_vector_cost as NULL, so determine_suggested_unroll_factor bails
> >> out early and reduction loops are never multi-accumulator-unrolled on
> >> these cores.  Fill it in from the ampere1 optimization guide (L1D two
> >> 128-bit loads plus one store per cycle, four general ops per cycle).
> >>
> >> gcc/ChangeLog:
> >>
> >>        * config/aarch64/tuning_models/ampere1.h (ampere1_scalar_issue_info)
> >>        (ampere1_advsimd_issue_info, ampere1_vec_issue_info): New.
> >>        (ampere1_vector_cost): Use ampere1_vec_issue_info for issue_info.
> >>        * config/aarch64/tuning_models/ampere1b.h (ampere1b_scalar_issue_info)
> >>        (ampere1b_advsimd_issue_info, ampere1b_vec_issue_info): New.
> >>        (ampere1b_vector_cost): Use ampere1b_vec_issue_info for issue_info.
> >>
> >> gcc/testsuite/ChangeLog:
> >>
> >>        * gcc.target/aarch64/ampere1-reduction-unroll-1.c: New test.
> >
> > LGTM.
> >
> >>
> >> Signed-off-by: Philipp Tomsich <philipp.tomsich@vrull.eu>
> >> ---
> >>
> >> gcc/config/aarch64/tuning_models/ampere1.h    | 32 ++++++++++++++++++-
> >> gcc/config/aarch64/tuning_models/ampere1b.h   | 32 ++++++++++++++++++-
> >> .../aarch64/ampere1-reduction-unroll-1.c      | 20 ++++++++++++
> >> 3 files changed, 82 insertions(+), 2 deletions(-)
> >> create mode 100644 gcc/testsuite/gcc.target/aarch64/ampere1-reduction-unroll-1.c
> >>
> >> diff --git a/gcc/config/aarch64/tuning_models/ampere1.h b/gcc/config/aarch64/tuning_models/ampere1.h
> >> index 53d45761b42f..e2a42f585dd9 100644
> >> --- a/gcc/config/aarch64/tuning_models/ampere1.h
> >> +++ b/gcc/config/aarch64/tuning_models/ampere1.h
> >> @@ -46,6 +46,36 @@ static const advsimd_vec_cost ampere1_advsimd_vector_cost =
> >>   1  /* store_cost  */
> >> };
> >>
> >> +static const aarch64_scalar_vec_issue_info ampere1_scalar_issue_info =
> >> +{
> >> +  3, /* loads_stores_per_cycle  */
> >> +  1, /* stores_per_cycle  */
> >> +  4, /* general_ops_per_cycle  */
> >> +  0, /* fp_simd_load_general_ops  */
> >> +  1 /* fp_simd_store_general_ops  */
> >> +};
> >> +
> >> +static const aarch64_advsimd_vec_issue_info ampere1_advsimd_issue_info =
> >> +{
> >> +  {
> >> +    2, /* loads_stores_per_cycle  */
> >> +    2, /* stores_per_cycle  */
> >> +    4, /* general_ops_per_cycle  */
> >> +    0, /* fp_simd_load_general_ops  */
> >> +    1 /* fp_simd_store_general_ops  */
> >> +  },
> >> +  2, /* ld2_st2_general_ops  */
> >> +  2, /* ld3_st3_general_ops  */
> >> +  3 /* ld4_st4_general_ops  */
> >> +};
> >> +
> >> +static const aarch64_vec_issue_info ampere1_vec_issue_info =
> >> +{
> >> +  &ampere1_scalar_issue_info,
> >> +  &ampere1_advsimd_issue_info,
> >> +  nullptr /* sve  */
> >> +};
> >> +
> >> /* Ampere-1 costs for vector insn classes.  */
> >> static const struct cpu_vector_cost ampere1_vector_cost =
> >> {
> >> @@ -57,7 +87,7 @@ static const struct cpu_vector_cost ampere1_vector_cost =
> >>   1, /* cond_not_taken_branch_cost  */
> >>   &ampere1_advsimd_vector_cost, /* advsimd  */
> >>   nullptr, /* sve  */
> >> -  nullptr  /* issue_info  */
> >> +  &ampere1_vec_issue_info /* issue_info  */
> >> };
> >>
> >> static const cpu_prefetch_tune ampere1_prefetch_tune =
> >> diff --git a/gcc/config/aarch64/tuning_models/ampere1b.h b/gcc/config/aarch64/tuning_models/ampere1b.h
> >> index b84713b53668..8ba434d27bbf 100644
> >> --- a/gcc/config/aarch64/tuning_models/ampere1b.h
> >> +++ b/gcc/config/aarch64/tuning_models/ampere1b.h
> >> @@ -57,6 +57,36 @@ static const advsimd_vec_cost ampere1b_advsimd_vector_cost =
> >>   1  /* store_cost  */
> >> };
> >>
> >> +static const aarch64_scalar_vec_issue_info ampere1b_scalar_issue_info =
> >> +{
> >> +  3, /* loads_stores_per_cycle  */
> >> +  1, /* stores_per_cycle  */
> >> +  4, /* general_ops_per_cycle  */
> >> +  0, /* fp_simd_load_general_ops  */
> >> +  1 /* fp_simd_store_general_ops  */
> >> +};
> >> +
> >> +static const aarch64_advsimd_vec_issue_info ampere1b_advsimd_issue_info =
> >> +{
> >> +  {
> >> +    2, /* loads_stores_per_cycle  */
> >> +    2, /* stores_per_cycle  */
> >> +    4, /* general_ops_per_cycle  */
> >> +    0, /* fp_simd_load_general_ops  */
> >> +    1 /* fp_simd_store_general_ops  */
> >> +  },
> >> +  2, /* ld2_st2_general_ops  */
> >> +  2, /* ld3_st3_general_ops  */
> >> +  3 /* ld4_st4_general_ops  */
> >> +};
> >> +
> >> +static const aarch64_vec_issue_info ampere1b_vec_issue_info =
> >> +{
> >> +  &ampere1b_scalar_issue_info,
> >> +  &ampere1b_advsimd_issue_info,
> >> +  nullptr /* sve  */
> >> +};
> >> +
> >> /* Ampere-1B costs for vector insn classes.  */
> >> static const struct cpu_vector_cost ampere1b_vector_cost =
> >> {
> >> @@ -68,7 +98,7 @@ static const struct cpu_vector_cost ampere1b_vector_cost =
> >>   1, /* cond_not_taken_branch_cost  */
> >>   &ampere1b_advsimd_vector_cost, /* advsimd  */
> >>   nullptr, /* sve  */
> >> -  nullptr  /* issue_info  */
> >> +  &ampere1b_vec_issue_info /* issue_info  */
> >> };
> >>
> >> static const struct tune_params ampere1b_tunings =
> >> diff --git a/gcc/testsuite/gcc.target/aarch64/ampere1-reduction-unroll-1.c b/gcc/testsuite/gcc.target/aarch64/ampere1-reduction-unroll-1.c
> >> new file mode 100644
> >> index 000000000000..61ab57304894
> >> --- /dev/null
> >> +++ b/gcc/testsuite/gcc.target/aarch64/ampere1-reduction-unroll-1.c
> >> @@ -0,0 +1,20 @@
> >> +/* { dg-do compile } */
> >> +/* { dg-options "-Ofast -mcpu=ampere1" } */
> >> +
> >> +/* The ampere1 family used a nullptr for the vector issue-info struct, so
> >> +   aarch64_vector_costs::determine_suggested_unroll_factor returned 1 at the
> >> +   `if (!vec_issue) return 1;' guard and reduction loops were never
> >> +   multi-accumulator-unrolled.  With issue_info populated, a unit-stride FP
> >> +   dot product widens to a suggested unroll factor of 4 and emits four
> >> +   independent `fmla v.4s' accumulator chains in the inner loop.  */
>
> I don’t find it useful for comments to talk about what the code used to be, they should describe what the code does now. I suggest you greatly simplify this comment and just say that for ampere1 you expect this loop to have unroll factor 4, or something to that effect.
> Thanks,
> Kyrill
>
>
> >> +
> >> +float
> >> +dot (const float *a, const float *b, int n)
> >> +{
> >> +  float s = 0.0f;
> >> +  for (int i = 0; i < n; i++)
> >> +    s += a[i] * b[i];
> >> +  return s;
> >> +}
> >> +
> >> +/* { dg-final { scan-assembler-times {\tfmla\tv[0-9]+\.4s} 4 } } */
> >> --
> >> 2.34.1
> >>
> >> base-commit: 3b443e764144526748d2b51d017d745f1956377e
>
>
  

Patch

diff --git a/gcc/config/aarch64/tuning_models/ampere1.h b/gcc/config/aarch64/tuning_models/ampere1.h
index 53d45761b42f..e2a42f585dd9 100644
--- a/gcc/config/aarch64/tuning_models/ampere1.h
+++ b/gcc/config/aarch64/tuning_models/ampere1.h
@@ -46,6 +46,36 @@  static const advsimd_vec_cost ampere1_advsimd_vector_cost =
   1  /* store_cost  */
 };
 
+static const aarch64_scalar_vec_issue_info ampere1_scalar_issue_info =
+{
+  3, /* loads_stores_per_cycle  */
+  1, /* stores_per_cycle  */
+  4, /* general_ops_per_cycle  */
+  0, /* fp_simd_load_general_ops  */
+  1 /* fp_simd_store_general_ops  */
+};
+
+static const aarch64_advsimd_vec_issue_info ampere1_advsimd_issue_info =
+{
+  {
+    2, /* loads_stores_per_cycle  */
+    2, /* stores_per_cycle  */
+    4, /* general_ops_per_cycle  */
+    0, /* fp_simd_load_general_ops  */
+    1 /* fp_simd_store_general_ops  */
+  },
+  2, /* ld2_st2_general_ops  */
+  2, /* ld3_st3_general_ops  */
+  3 /* ld4_st4_general_ops  */
+};
+
+static const aarch64_vec_issue_info ampere1_vec_issue_info =
+{
+  &ampere1_scalar_issue_info,
+  &ampere1_advsimd_issue_info,
+  nullptr /* sve  */
+};
+
 /* Ampere-1 costs for vector insn classes.  */
 static const struct cpu_vector_cost ampere1_vector_cost =
 {
@@ -57,7 +87,7 @@  static const struct cpu_vector_cost ampere1_vector_cost =
   1, /* cond_not_taken_branch_cost  */
   &ampere1_advsimd_vector_cost, /* advsimd  */
   nullptr, /* sve  */
-  nullptr  /* issue_info  */
+  &ampere1_vec_issue_info /* issue_info  */
 };
 
 static const cpu_prefetch_tune ampere1_prefetch_tune =
diff --git a/gcc/config/aarch64/tuning_models/ampere1b.h b/gcc/config/aarch64/tuning_models/ampere1b.h
index b84713b53668..8ba434d27bbf 100644
--- a/gcc/config/aarch64/tuning_models/ampere1b.h
+++ b/gcc/config/aarch64/tuning_models/ampere1b.h
@@ -57,6 +57,36 @@  static const advsimd_vec_cost ampere1b_advsimd_vector_cost =
   1  /* store_cost  */
 };
 
+static const aarch64_scalar_vec_issue_info ampere1b_scalar_issue_info =
+{
+  3, /* loads_stores_per_cycle  */
+  1, /* stores_per_cycle  */
+  4, /* general_ops_per_cycle  */
+  0, /* fp_simd_load_general_ops  */
+  1 /* fp_simd_store_general_ops  */
+};
+
+static const aarch64_advsimd_vec_issue_info ampere1b_advsimd_issue_info =
+{
+  {
+    2, /* loads_stores_per_cycle  */
+    2, /* stores_per_cycle  */
+    4, /* general_ops_per_cycle  */
+    0, /* fp_simd_load_general_ops  */
+    1 /* fp_simd_store_general_ops  */
+  },
+  2, /* ld2_st2_general_ops  */
+  2, /* ld3_st3_general_ops  */
+  3 /* ld4_st4_general_ops  */
+};
+
+static const aarch64_vec_issue_info ampere1b_vec_issue_info =
+{
+  &ampere1b_scalar_issue_info,
+  &ampere1b_advsimd_issue_info,
+  nullptr /* sve  */
+};
+
 /* Ampere-1B costs for vector insn classes.  */
 static const struct cpu_vector_cost ampere1b_vector_cost =
 {
@@ -68,7 +98,7 @@  static const struct cpu_vector_cost ampere1b_vector_cost =
   1, /* cond_not_taken_branch_cost  */
   &ampere1b_advsimd_vector_cost, /* advsimd  */
   nullptr, /* sve  */
-  nullptr  /* issue_info  */
+  &ampere1b_vec_issue_info /* issue_info  */
 };
 
 static const struct tune_params ampere1b_tunings =
diff --git a/gcc/testsuite/gcc.target/aarch64/ampere1-reduction-unroll-1.c b/gcc/testsuite/gcc.target/aarch64/ampere1-reduction-unroll-1.c
new file mode 100644
index 000000000000..61ab57304894
--- /dev/null
+++ b/gcc/testsuite/gcc.target/aarch64/ampere1-reduction-unroll-1.c
@@ -0,0 +1,20 @@ 
+/* { dg-do compile } */
+/* { dg-options "-Ofast -mcpu=ampere1" } */
+
+/* The ampere1 family used a nullptr for the vector issue-info struct, so
+   aarch64_vector_costs::determine_suggested_unroll_factor returned 1 at the
+   `if (!vec_issue) return 1;' guard and reduction loops were never
+   multi-accumulator-unrolled.  With issue_info populated, a unit-stride FP
+   dot product widens to a suggested unroll factor of 4 and emits four
+   independent `fmla v.4s' accumulator chains in the inner loop.  */
+
+float
+dot (const float *a, const float *b, int n)
+{
+  float s = 0.0f;
+  for (int i = 0; i < n; i++)
+    s += a[i] * b[i];
+  return s;
+}
+
+/* { dg-final { scan-assembler-times {\tfmla\tv[0-9]+\.4s} 4 } } */