[gcc(refs/users/mikael/heads/refactor_descriptor_v08)] Régénération fichiers générés
Mikael Morin
mikael@gcc.gnu.org
Fri Sep 12 18:02:00 GMT 2025
https://gcc.gnu.org/g:722d3c13f724ee5436e6d39b0fb70e3ddbfdf534
commit 722d3c13f724ee5436e6d39b0fb70e3ddbfdf534
Author: Mikael Morin <mikael@gcc.gnu.org>
Date: Fri Sep 12 14:49:41 2025 +0200
Régénération fichiers générés
Régénération fichiers générés
Régénération fichiers générés
Régénération fichiers générés
Régénération fichiers générés
Diff:
---
libgfortran/generated/bessel_r10.c | 34 +-
libgfortran/generated/bessel_r16.c | 34 +-
libgfortran/generated/bessel_r17.c | 34 +-
libgfortran/generated/bessel_r4.c | 34 +-
libgfortran/generated/bessel_r8.c | 34 +-
libgfortran/generated/cshift0_c10.c | 2 +-
libgfortran/generated/cshift0_c16.c | 2 +-
libgfortran/generated/cshift0_c17.c | 2 +-
libgfortran/generated/cshift0_c4.c | 2 +-
libgfortran/generated/cshift0_c8.c | 2 +-
libgfortran/generated/cshift0_i1.c | 2 +-
libgfortran/generated/cshift0_i16.c | 2 +-
libgfortran/generated/cshift0_i2.c | 2 +-
libgfortran/generated/cshift0_i4.c | 2 +-
libgfortran/generated/cshift0_i8.c | 2 +-
libgfortran/generated/cshift0_r10.c | 2 +-
libgfortran/generated/cshift0_r16.c | 2 +-
libgfortran/generated/cshift0_r17.c | 2 +-
libgfortran/generated/cshift0_r4.c | 2 +-
libgfortran/generated/cshift0_r8.c | 2 +-
libgfortran/generated/cshift1_16.c | 2 +-
libgfortran/generated/cshift1_16_c10.c | 2 +-
libgfortran/generated/cshift1_16_c16.c | 2 +-
libgfortran/generated/cshift1_16_c17.c | 2 +-
libgfortran/generated/cshift1_16_c4.c | 2 +-
libgfortran/generated/cshift1_16_c8.c | 2 +-
libgfortran/generated/cshift1_16_i1.c | 2 +-
libgfortran/generated/cshift1_16_i16.c | 2 +-
libgfortran/generated/cshift1_16_i2.c | 2 +-
libgfortran/generated/cshift1_16_i4.c | 2 +-
libgfortran/generated/cshift1_16_i8.c | 2 +-
libgfortran/generated/cshift1_16_r10.c | 2 +-
libgfortran/generated/cshift1_16_r16.c | 2 +-
libgfortran/generated/cshift1_16_r17.c | 2 +-
libgfortran/generated/cshift1_16_r4.c | 2 +-
libgfortran/generated/cshift1_16_r8.c | 2 +-
libgfortran/generated/cshift1_4.c | 2 +-
libgfortran/generated/cshift1_4_c10.c | 2 +-
libgfortran/generated/cshift1_4_c16.c | 2 +-
libgfortran/generated/cshift1_4_c17.c | 2 +-
libgfortran/generated/cshift1_4_c4.c | 2 +-
libgfortran/generated/cshift1_4_c8.c | 2 +-
libgfortran/generated/cshift1_4_i1.c | 2 +-
libgfortran/generated/cshift1_4_i16.c | 2 +-
libgfortran/generated/cshift1_4_i2.c | 2 +-
libgfortran/generated/cshift1_4_i4.c | 2 +-
libgfortran/generated/cshift1_4_i8.c | 2 +-
libgfortran/generated/cshift1_4_r10.c | 2 +-
libgfortran/generated/cshift1_4_r16.c | 2 +-
libgfortran/generated/cshift1_4_r17.c | 2 +-
libgfortran/generated/cshift1_4_r4.c | 2 +-
libgfortran/generated/cshift1_4_r8.c | 2 +-
libgfortran/generated/cshift1_8.c | 2 +-
libgfortran/generated/cshift1_8_c10.c | 2 +-
libgfortran/generated/cshift1_8_c16.c | 2 +-
libgfortran/generated/cshift1_8_c17.c | 2 +-
libgfortran/generated/cshift1_8_c4.c | 2 +-
libgfortran/generated/cshift1_8_c8.c | 2 +-
libgfortran/generated/cshift1_8_i1.c | 2 +-
libgfortran/generated/cshift1_8_i16.c | 2 +-
libgfortran/generated/cshift1_8_i2.c | 2 +-
libgfortran/generated/cshift1_8_i4.c | 2 +-
libgfortran/generated/cshift1_8_i8.c | 2 +-
libgfortran/generated/cshift1_8_r10.c | 2 +-
libgfortran/generated/cshift1_8_r16.c | 2 +-
libgfortran/generated/cshift1_8_r17.c | 2 +-
libgfortran/generated/cshift1_8_r4.c | 2 +-
libgfortran/generated/cshift1_8_r8.c | 2 +-
libgfortran/generated/findloc0_c10.c | 32 +-
libgfortran/generated/findloc0_c16.c | 32 +-
libgfortran/generated/findloc0_c17.c | 32 +-
libgfortran/generated/findloc0_c4.c | 32 +-
libgfortran/generated/findloc0_c8.c | 32 +-
libgfortran/generated/findloc0_i1.c | 32 +-
libgfortran/generated/findloc0_i16.c | 32 +-
libgfortran/generated/findloc0_i2.c | 32 +-
libgfortran/generated/findloc0_i4.c | 32 +-
libgfortran/generated/findloc0_i8.c | 32 +-
libgfortran/generated/findloc0_r10.c | 32 +-
libgfortran/generated/findloc0_r16.c | 32 +-
libgfortran/generated/findloc0_r17.c | 32 +-
libgfortran/generated/findloc0_r4.c | 32 +-
libgfortran/generated/findloc0_r8.c | 32 +-
libgfortran/generated/findloc0_s1.c | 32 +-
libgfortran/generated/findloc0_s4.c | 32 +-
libgfortran/generated/findloc1_c10.c | 4 +-
libgfortran/generated/findloc1_c16.c | 4 +-
libgfortran/generated/findloc1_c17.c | 4 +-
libgfortran/generated/findloc1_c4.c | 4 +-
libgfortran/generated/findloc1_c8.c | 4 +-
libgfortran/generated/findloc1_i1.c | 4 +-
libgfortran/generated/findloc1_i16.c | 4 +-
libgfortran/generated/findloc1_i2.c | 4 +-
libgfortran/generated/findloc1_i4.c | 4 +-
libgfortran/generated/findloc1_i8.c | 4 +-
libgfortran/generated/findloc1_r10.c | 4 +-
libgfortran/generated/findloc1_r16.c | 4 +-
libgfortran/generated/findloc1_r17.c | 4 +-
libgfortran/generated/findloc1_r4.c | 4 +-
libgfortran/generated/findloc1_r8.c | 4 +-
libgfortran/generated/findloc1_s1.c | 4 +-
libgfortran/generated/findloc1_s4.c | 4 +-
libgfortran/generated/findloc2_s1.c | 4 +-
libgfortran/generated/findloc2_s4.c | 4 +-
libgfortran/generated/matmul_c10.c | 1175 ++++++++++++++++--------------
libgfortran/generated/matmul_c16.c | 1175 ++++++++++++++++--------------
libgfortran/generated/matmul_c17.c | 1175 ++++++++++++++++--------------
libgfortran/generated/matmul_c4.c | 1175 ++++++++++++++++--------------
libgfortran/generated/matmul_c8.c | 1175 ++++++++++++++++--------------
libgfortran/generated/matmul_i1.c | 1175 ++++++++++++++++--------------
libgfortran/generated/matmul_i16.c | 1175 ++++++++++++++++--------------
libgfortran/generated/matmul_i2.c | 1175 ++++++++++++++++--------------
libgfortran/generated/matmul_i4.c | 1175 ++++++++++++++++--------------
libgfortran/generated/matmul_i8.c | 1175 ++++++++++++++++--------------
libgfortran/generated/matmul_r10.c | 1175 ++++++++++++++++--------------
libgfortran/generated/matmul_r16.c | 1175 ++++++++++++++++--------------
libgfortran/generated/matmul_r17.c | 1175 ++++++++++++++++--------------
libgfortran/generated/matmul_r4.c | 1175 ++++++++++++++++--------------
libgfortran/generated/matmul_r8.c | 1175 ++++++++++++++++--------------
libgfortran/generated/matmulavx128_c10.c | 470 ++++++------
libgfortran/generated/matmulavx128_c16.c | 470 ++++++------
libgfortran/generated/matmulavx128_c17.c | 470 ++++++------
libgfortran/generated/matmulavx128_c4.c | 470 ++++++------
libgfortran/generated/matmulavx128_c8.c | 470 ++++++------
libgfortran/generated/matmulavx128_i1.c | 470 ++++++------
libgfortran/generated/matmulavx128_i16.c | 470 ++++++------
libgfortran/generated/matmulavx128_i2.c | 470 ++++++------
libgfortran/generated/matmulavx128_i4.c | 470 ++++++------
libgfortran/generated/matmulavx128_i8.c | 470 ++++++------
libgfortran/generated/matmulavx128_r10.c | 470 ++++++------
libgfortran/generated/matmulavx128_r16.c | 470 ++++++------
libgfortran/generated/matmulavx128_r17.c | 470 ++++++------
libgfortran/generated/matmulavx128_r4.c | 470 ++++++------
libgfortran/generated/matmulavx128_r8.c | 470 ++++++------
libgfortran/generated/maxloc0_16_i1.c | 38 +-
libgfortran/generated/maxloc0_16_i16.c | 38 +-
libgfortran/generated/maxloc0_16_i2.c | 38 +-
libgfortran/generated/maxloc0_16_i4.c | 38 +-
libgfortran/generated/maxloc0_16_i8.c | 38 +-
libgfortran/generated/maxloc0_16_m1.c | 38 +-
libgfortran/generated/maxloc0_16_m16.c | 38 +-
libgfortran/generated/maxloc0_16_m2.c | 38 +-
libgfortran/generated/maxloc0_16_m4.c | 38 +-
libgfortran/generated/maxloc0_16_m8.c | 38 +-
libgfortran/generated/maxloc0_16_r10.c | 38 +-
libgfortran/generated/maxloc0_16_r16.c | 38 +-
libgfortran/generated/maxloc0_16_r17.c | 38 +-
libgfortran/generated/maxloc0_16_r4.c | 38 +-
libgfortran/generated/maxloc0_16_r8.c | 38 +-
libgfortran/generated/maxloc0_16_s1.c | 26 +-
libgfortran/generated/maxloc0_16_s4.c | 26 +-
libgfortran/generated/maxloc0_4_i1.c | 38 +-
libgfortran/generated/maxloc0_4_i16.c | 38 +-
libgfortran/generated/maxloc0_4_i2.c | 38 +-
libgfortran/generated/maxloc0_4_i4.c | 38 +-
libgfortran/generated/maxloc0_4_i8.c | 38 +-
libgfortran/generated/maxloc0_4_m1.c | 38 +-
libgfortran/generated/maxloc0_4_m16.c | 38 +-
libgfortran/generated/maxloc0_4_m2.c | 38 +-
libgfortran/generated/maxloc0_4_m4.c | 38 +-
libgfortran/generated/maxloc0_4_m8.c | 38 +-
libgfortran/generated/maxloc0_4_r10.c | 38 +-
libgfortran/generated/maxloc0_4_r16.c | 38 +-
libgfortran/generated/maxloc0_4_r17.c | 38 +-
libgfortran/generated/maxloc0_4_r4.c | 38 +-
libgfortran/generated/maxloc0_4_r8.c | 38 +-
libgfortran/generated/maxloc0_4_s1.c | 26 +-
libgfortran/generated/maxloc0_4_s4.c | 26 +-
libgfortran/generated/maxloc0_8_i1.c | 38 +-
libgfortran/generated/maxloc0_8_i16.c | 38 +-
libgfortran/generated/maxloc0_8_i2.c | 38 +-
libgfortran/generated/maxloc0_8_i4.c | 38 +-
libgfortran/generated/maxloc0_8_i8.c | 38 +-
libgfortran/generated/maxloc0_8_m1.c | 38 +-
libgfortran/generated/maxloc0_8_m16.c | 38 +-
libgfortran/generated/maxloc0_8_m2.c | 38 +-
libgfortran/generated/maxloc0_8_m4.c | 38 +-
libgfortran/generated/maxloc0_8_m8.c | 38 +-
libgfortran/generated/maxloc0_8_r10.c | 38 +-
libgfortran/generated/maxloc0_8_r16.c | 38 +-
libgfortran/generated/maxloc0_8_r17.c | 38 +-
libgfortran/generated/maxloc0_8_r4.c | 38 +-
libgfortran/generated/maxloc0_8_r8.c | 38 +-
libgfortran/generated/maxloc0_8_s1.c | 26 +-
libgfortran/generated/maxloc0_8_s4.c | 26 +-
libgfortran/generated/maxloc2_16_s1.c | 2 +-
libgfortran/generated/maxloc2_16_s4.c | 2 +-
libgfortran/generated/maxloc2_4_s1.c | 2 +-
libgfortran/generated/maxloc2_4_s4.c | 2 +-
libgfortran/generated/maxloc2_8_s1.c | 2 +-
libgfortran/generated/maxloc2_8_s4.c | 2 +-
libgfortran/generated/minloc0_16_i1.c | 38 +-
libgfortran/generated/minloc0_16_i16.c | 38 +-
libgfortran/generated/minloc0_16_i2.c | 38 +-
libgfortran/generated/minloc0_16_i4.c | 38 +-
libgfortran/generated/minloc0_16_i8.c | 38 +-
libgfortran/generated/minloc0_16_m1.c | 38 +-
libgfortran/generated/minloc0_16_m16.c | 38 +-
libgfortran/generated/minloc0_16_m2.c | 38 +-
libgfortran/generated/minloc0_16_m4.c | 38 +-
libgfortran/generated/minloc0_16_m8.c | 38 +-
libgfortran/generated/minloc0_16_r10.c | 38 +-
libgfortran/generated/minloc0_16_r16.c | 38 +-
libgfortran/generated/minloc0_16_r17.c | 38 +-
libgfortran/generated/minloc0_16_r4.c | 38 +-
libgfortran/generated/minloc0_16_r8.c | 38 +-
libgfortran/generated/minloc0_16_s1.c | 26 +-
libgfortran/generated/minloc0_16_s4.c | 26 +-
libgfortran/generated/minloc0_4_i1.c | 38 +-
libgfortran/generated/minloc0_4_i16.c | 38 +-
libgfortran/generated/minloc0_4_i2.c | 38 +-
libgfortran/generated/minloc0_4_i4.c | 38 +-
libgfortran/generated/minloc0_4_i8.c | 38 +-
libgfortran/generated/minloc0_4_m1.c | 38 +-
libgfortran/generated/minloc0_4_m16.c | 38 +-
libgfortran/generated/minloc0_4_m2.c | 38 +-
libgfortran/generated/minloc0_4_m4.c | 38 +-
libgfortran/generated/minloc0_4_m8.c | 38 +-
libgfortran/generated/minloc0_4_r10.c | 38 +-
libgfortran/generated/minloc0_4_r16.c | 38 +-
libgfortran/generated/minloc0_4_r17.c | 38 +-
libgfortran/generated/minloc0_4_r4.c | 38 +-
libgfortran/generated/minloc0_4_r8.c | 38 +-
libgfortran/generated/minloc0_4_s1.c | 26 +-
libgfortran/generated/minloc0_4_s4.c | 26 +-
libgfortran/generated/minloc0_8_i1.c | 38 +-
libgfortran/generated/minloc0_8_i16.c | 38 +-
libgfortran/generated/minloc0_8_i2.c | 38 +-
libgfortran/generated/minloc0_8_i4.c | 38 +-
libgfortran/generated/minloc0_8_i8.c | 38 +-
libgfortran/generated/minloc0_8_m1.c | 38 +-
libgfortran/generated/minloc0_8_m16.c | 38 +-
libgfortran/generated/minloc0_8_m2.c | 38 +-
libgfortran/generated/minloc0_8_m4.c | 38 +-
libgfortran/generated/minloc0_8_m8.c | 38 +-
libgfortran/generated/minloc0_8_r10.c | 38 +-
libgfortran/generated/minloc0_8_r16.c | 38 +-
libgfortran/generated/minloc0_8_r17.c | 38 +-
libgfortran/generated/minloc0_8_r4.c | 38 +-
libgfortran/generated/minloc0_8_r8.c | 38 +-
libgfortran/generated/minloc0_8_s1.c | 26 +-
libgfortran/generated/minloc0_8_s4.c | 26 +-
libgfortran/generated/minloc2_16_s1.c | 2 +-
libgfortran/generated/minloc2_16_s4.c | 2 +-
libgfortran/generated/minloc2_4_s1.c | 2 +-
libgfortran/generated/minloc2_4_s4.c | 2 +-
libgfortran/generated/minloc2_8_s1.c | 2 +-
libgfortran/generated/minloc2_8_s4.c | 2 +-
libgfortran/generated/pack_c10.c | 2 +-
libgfortran/generated/pack_c16.c | 2 +-
libgfortran/generated/pack_c17.c | 2 +-
libgfortran/generated/pack_c4.c | 2 +-
libgfortran/generated/pack_c8.c | 2 +-
libgfortran/generated/pack_i1.c | 2 +-
libgfortran/generated/pack_i16.c | 2 +-
libgfortran/generated/pack_i2.c | 2 +-
libgfortran/generated/pack_i4.c | 2 +-
libgfortran/generated/pack_i8.c | 2 +-
libgfortran/generated/pack_r10.c | 2 +-
libgfortran/generated/pack_r16.c | 2 +-
libgfortran/generated/pack_r17.c | 2 +-
libgfortran/generated/pack_r4.c | 2 +-
libgfortran/generated/pack_r8.c | 2 +-
libgfortran/generated/reshape_c10.c | 4 +-
libgfortran/generated/reshape_c16.c | 4 +-
libgfortran/generated/reshape_c17.c | 4 +-
libgfortran/generated/reshape_c4.c | 4 +-
libgfortran/generated/reshape_c8.c | 4 +-
libgfortran/generated/reshape_i16.c | 4 +-
libgfortran/generated/reshape_i4.c | 4 +-
libgfortran/generated/reshape_i8.c | 4 +-
libgfortran/generated/reshape_r10.c | 4 +-
libgfortran/generated/reshape_r16.c | 4 +-
libgfortran/generated/reshape_r17.c | 4 +-
libgfortran/generated/reshape_r4.c | 4 +-
libgfortran/generated/reshape_r8.c | 4 +-
libgfortran/generated/shape_i1.c | 5 +-
libgfortran/generated/shape_i16.c | 5 +-
libgfortran/generated/shape_i2.c | 5 +-
libgfortran/generated/shape_i4.c | 5 +-
libgfortran/generated/shape_i8.c | 5 +-
281 files changed, 14856 insertions(+), 14598 deletions(-)
diff --git a/libgfortran/generated/bessel_r10.c b/libgfortran/generated/bessel_r10.c
index 2b38e4c2fddb..07d378af6cdf 100644
--- a/libgfortran/generated/bessel_r10.c
+++ b/libgfortran/generated/bessel_r10.c
@@ -43,12 +43,9 @@ void
bessel_jn_r10 (gfc_array_r10 * const restrict ret, int n1, int n2, GFC_REAL_10 x)
{
int i;
- index_type stride;
GFC_REAL_10 last1, last2, x2rev;
- stride = GFC_DESCRIPTOR_STRIDE(ret,0);
-
if (ret->base_addr == NULL)
{
size_t size = n2 < n1 ? 0 : n2-n1+1;
@@ -66,24 +63,22 @@ bessel_jn_r10 (gfc_array_r10 * const restrict ret, int n1, int n2, GFC_REAL_10 x
"(%ld vs. %ld)", (long int) n2-n1,
(long int) GFC_DESCRIPTOR_EXTENT(ret,0));
- stride = GFC_DESCRIPTOR_STRIDE(ret,0);
-
if (unlikely (x == 0))
{
ret->base_addr[0] = 1;
for (i = 1; i <= n2-n1; i++)
- ret->base_addr[i*stride] = 0;
+ GFC_DESCRIPTOR1_ELEM (ret, i) = 0;
return;
}
last1 = MATHFUNC(jn) (n2, x);
- ret->base_addr[(n2-n1)*stride] = last1;
+ GFC_DESCRIPTOR1_ELEM (ret, n2-n1) = last1;
if (n1 == n2)
return;
last2 = MATHFUNC(jn) (n2 - 1, x);
- ret->base_addr[(n2-n1-1)*stride] = last2;
+ GFC_DESCRIPTOR1_ELEM (ret, n2-n1-1) = last2;
if (n1 + 1 == n2)
return;
@@ -92,9 +87,9 @@ bessel_jn_r10 (gfc_array_r10 * const restrict ret, int n1, int n2, GFC_REAL_10 x
for (i = n2-n1-2; i >= 0; i--)
{
- ret->base_addr[i*stride] = x2rev * (i+1+n1) * last2 - last1;
+ GFC_DESCRIPTOR1_ELEM (ret, i) = x2rev * (i+1+n1) * last2 - last1;
last1 = last2;
- last2 = ret->base_addr[i*stride];
+ last2 = GFC_DESCRIPTOR1_ELEM (ret, i);
}
}
@@ -110,12 +105,9 @@ bessel_yn_r10 (gfc_array_r10 * const restrict ret, int n1, int n2,
GFC_REAL_10 x)
{
int i;
- index_type stride;
GFC_REAL_10 last1, last2, x2rev;
- stride = GFC_DESCRIPTOR_STRIDE(ret,0);
-
if (ret->base_addr == NULL)
{
size_t size = n2 < n1 ? 0 : n2-n1+1;
@@ -133,27 +125,25 @@ bessel_yn_r10 (gfc_array_r10 * const restrict ret, int n1, int n2,
"(%ld vs. %ld)", (long int) n2-n1,
(long int) GFC_DESCRIPTOR_EXTENT(ret,0));
- stride = GFC_DESCRIPTOR_STRIDE(ret,0);
-
if (unlikely (x == 0))
{
for (i = 0; i <= n2-n1; i++)
#if defined(GFC_REAL_10_INFINITY)
- ret->base_addr[i*stride] = -GFC_REAL_10_INFINITY;
+ GFC_DESCRIPTOR1_ELEM (ret, i) = -GFC_REAL_10_INFINITY;
#else
- ret->base_addr[i*stride] = -GFC_REAL_10_HUGE;
+ GFC_DESCRIPTOR1_ELEM (ret, i) = -GFC_REAL_10_HUGE;
#endif
return;
}
last1 = MATHFUNC(yn) (n1, x);
- ret->base_addr[0] = last1;
+ GFC_DESCRIPTOR1_ELEM (ret, 0) = last1;
if (n1 == n2)
return;
last2 = MATHFUNC(yn) (n1 + 1, x);
- ret->base_addr[1*stride] = last2;
+ GFC_DESCRIPTOR1_ELEM (ret, 1) = last2;
if (n1 + 1 == n2)
return;
@@ -165,14 +155,14 @@ bessel_yn_r10 (gfc_array_r10 * const restrict ret, int n1, int n2,
#if defined(GFC_REAL_10_INFINITY)
if (unlikely (last2 == -GFC_REAL_10_INFINITY))
{
- ret->base_addr[i*stride] = -GFC_REAL_10_INFINITY;
+ GFC_DESCRIPTOR1_ELEM (ret, i) = -GFC_REAL_10_INFINITY;
}
else
#endif
{
- ret->base_addr[i*stride] = x2rev * (i-1+n1) * last2 - last1;
+ GFC_DESCRIPTOR1_ELEM (ret, i) = x2rev * (i-1+n1) * last2 - last1;
last1 = last2;
- last2 = ret->base_addr[i*stride];
+ last2 = GFC_DESCRIPTOR1_ELEM (ret, i);
}
}
}
diff --git a/libgfortran/generated/bessel_r16.c b/libgfortran/generated/bessel_r16.c
index bf809eba26f6..fbb74ee115d7 100644
--- a/libgfortran/generated/bessel_r16.c
+++ b/libgfortran/generated/bessel_r16.c
@@ -51,12 +51,9 @@ void
bessel_jn_r16 (gfc_array_r16 * const restrict ret, int n1, int n2, GFC_REAL_16 x)
{
int i;
- index_type stride;
GFC_REAL_16 last1, last2, x2rev;
- stride = GFC_DESCRIPTOR_STRIDE(ret,0);
-
if (ret->base_addr == NULL)
{
size_t size = n2 < n1 ? 0 : n2-n1+1;
@@ -74,24 +71,22 @@ bessel_jn_r16 (gfc_array_r16 * const restrict ret, int n1, int n2, GFC_REAL_16 x
"(%ld vs. %ld)", (long int) n2-n1,
(long int) GFC_DESCRIPTOR_EXTENT(ret,0));
- stride = GFC_DESCRIPTOR_STRIDE(ret,0);
-
if (unlikely (x == 0))
{
ret->base_addr[0] = 1;
for (i = 1; i <= n2-n1; i++)
- ret->base_addr[i*stride] = 0;
+ GFC_DESCRIPTOR1_ELEM (ret, i) = 0;
return;
}
last1 = MATHFUNC(jn) (n2, x);
- ret->base_addr[(n2-n1)*stride] = last1;
+ GFC_DESCRIPTOR1_ELEM (ret, n2-n1) = last1;
if (n1 == n2)
return;
last2 = MATHFUNC(jn) (n2 - 1, x);
- ret->base_addr[(n2-n1-1)*stride] = last2;
+ GFC_DESCRIPTOR1_ELEM (ret, n2-n1-1) = last2;
if (n1 + 1 == n2)
return;
@@ -100,9 +95,9 @@ bessel_jn_r16 (gfc_array_r16 * const restrict ret, int n1, int n2, GFC_REAL_16 x
for (i = n2-n1-2; i >= 0; i--)
{
- ret->base_addr[i*stride] = x2rev * (i+1+n1) * last2 - last1;
+ GFC_DESCRIPTOR1_ELEM (ret, i) = x2rev * (i+1+n1) * last2 - last1;
last1 = last2;
- last2 = ret->base_addr[i*stride];
+ last2 = GFC_DESCRIPTOR1_ELEM (ret, i);
}
}
@@ -118,12 +113,9 @@ bessel_yn_r16 (gfc_array_r16 * const restrict ret, int n1, int n2,
GFC_REAL_16 x)
{
int i;
- index_type stride;
GFC_REAL_16 last1, last2, x2rev;
- stride = GFC_DESCRIPTOR_STRIDE(ret,0);
-
if (ret->base_addr == NULL)
{
size_t size = n2 < n1 ? 0 : n2-n1+1;
@@ -141,27 +133,25 @@ bessel_yn_r16 (gfc_array_r16 * const restrict ret, int n1, int n2,
"(%ld vs. %ld)", (long int) n2-n1,
(long int) GFC_DESCRIPTOR_EXTENT(ret,0));
- stride = GFC_DESCRIPTOR_STRIDE(ret,0);
-
if (unlikely (x == 0))
{
for (i = 0; i <= n2-n1; i++)
#if defined(GFC_REAL_16_INFINITY)
- ret->base_addr[i*stride] = -GFC_REAL_16_INFINITY;
+ GFC_DESCRIPTOR1_ELEM (ret, i) = -GFC_REAL_16_INFINITY;
#else
- ret->base_addr[i*stride] = -GFC_REAL_16_HUGE;
+ GFC_DESCRIPTOR1_ELEM (ret, i) = -GFC_REAL_16_HUGE;
#endif
return;
}
last1 = MATHFUNC(yn) (n1, x);
- ret->base_addr[0] = last1;
+ GFC_DESCRIPTOR1_ELEM (ret, 0) = last1;
if (n1 == n2)
return;
last2 = MATHFUNC(yn) (n1 + 1, x);
- ret->base_addr[1*stride] = last2;
+ GFC_DESCRIPTOR1_ELEM (ret, 1) = last2;
if (n1 + 1 == n2)
return;
@@ -173,14 +163,14 @@ bessel_yn_r16 (gfc_array_r16 * const restrict ret, int n1, int n2,
#if defined(GFC_REAL_16_INFINITY)
if (unlikely (last2 == -GFC_REAL_16_INFINITY))
{
- ret->base_addr[i*stride] = -GFC_REAL_16_INFINITY;
+ GFC_DESCRIPTOR1_ELEM (ret, i) = -GFC_REAL_16_INFINITY;
}
else
#endif
{
- ret->base_addr[i*stride] = x2rev * (i-1+n1) * last2 - last1;
+ GFC_DESCRIPTOR1_ELEM (ret, i) = x2rev * (i-1+n1) * last2 - last1;
last1 = last2;
- last2 = ret->base_addr[i*stride];
+ last2 = GFC_DESCRIPTOR1_ELEM (ret, i);
}
}
}
diff --git a/libgfortran/generated/bessel_r17.c b/libgfortran/generated/bessel_r17.c
index a9e9e9788164..27c398fabd28 100644
--- a/libgfortran/generated/bessel_r17.c
+++ b/libgfortran/generated/bessel_r17.c
@@ -49,12 +49,9 @@ void
bessel_jn_r17 (gfc_array_r17 * const restrict ret, int n1, int n2, GFC_REAL_17 x)
{
int i;
- index_type stride;
GFC_REAL_17 last1, last2, x2rev;
- stride = GFC_DESCRIPTOR_STRIDE(ret,0);
-
if (ret->base_addr == NULL)
{
size_t size = n2 < n1 ? 0 : n2-n1+1;
@@ -72,24 +69,22 @@ bessel_jn_r17 (gfc_array_r17 * const restrict ret, int n1, int n2, GFC_REAL_17 x
"(%ld vs. %ld)", (long int) n2-n1,
(long int) GFC_DESCRIPTOR_EXTENT(ret,0));
- stride = GFC_DESCRIPTOR_STRIDE(ret,0);
-
if (unlikely (x == 0))
{
ret->base_addr[0] = 1;
for (i = 1; i <= n2-n1; i++)
- ret->base_addr[i*stride] = 0;
+ GFC_DESCRIPTOR1_ELEM (ret, i) = 0;
return;
}
last1 = MATHFUNC(jn) (n2, x);
- ret->base_addr[(n2-n1)*stride] = last1;
+ GFC_DESCRIPTOR1_ELEM (ret, n2-n1) = last1;
if (n1 == n2)
return;
last2 = MATHFUNC(jn) (n2 - 1, x);
- ret->base_addr[(n2-n1-1)*stride] = last2;
+ GFC_DESCRIPTOR1_ELEM (ret, n2-n1-1) = last2;
if (n1 + 1 == n2)
return;
@@ -98,9 +93,9 @@ bessel_jn_r17 (gfc_array_r17 * const restrict ret, int n1, int n2, GFC_REAL_17 x
for (i = n2-n1-2; i >= 0; i--)
{
- ret->base_addr[i*stride] = x2rev * (i+1+n1) * last2 - last1;
+ GFC_DESCRIPTOR1_ELEM (ret, i) = x2rev * (i+1+n1) * last2 - last1;
last1 = last2;
- last2 = ret->base_addr[i*stride];
+ last2 = GFC_DESCRIPTOR1_ELEM (ret, i);
}
}
@@ -116,12 +111,9 @@ bessel_yn_r17 (gfc_array_r17 * const restrict ret, int n1, int n2,
GFC_REAL_17 x)
{
int i;
- index_type stride;
GFC_REAL_17 last1, last2, x2rev;
- stride = GFC_DESCRIPTOR_STRIDE(ret,0);
-
if (ret->base_addr == NULL)
{
size_t size = n2 < n1 ? 0 : n2-n1+1;
@@ -139,27 +131,25 @@ bessel_yn_r17 (gfc_array_r17 * const restrict ret, int n1, int n2,
"(%ld vs. %ld)", (long int) n2-n1,
(long int) GFC_DESCRIPTOR_EXTENT(ret,0));
- stride = GFC_DESCRIPTOR_STRIDE(ret,0);
-
if (unlikely (x == 0))
{
for (i = 0; i <= n2-n1; i++)
#if defined(GFC_REAL_17_INFINITY)
- ret->base_addr[i*stride] = -GFC_REAL_17_INFINITY;
+ GFC_DESCRIPTOR1_ELEM (ret, i) = -GFC_REAL_17_INFINITY;
#else
- ret->base_addr[i*stride] = -GFC_REAL_17_HUGE;
+ GFC_DESCRIPTOR1_ELEM (ret, i) = -GFC_REAL_17_HUGE;
#endif
return;
}
last1 = MATHFUNC(yn) (n1, x);
- ret->base_addr[0] = last1;
+ GFC_DESCRIPTOR1_ELEM (ret, 0) = last1;
if (n1 == n2)
return;
last2 = MATHFUNC(yn) (n1 + 1, x);
- ret->base_addr[1*stride] = last2;
+ GFC_DESCRIPTOR1_ELEM (ret, 1) = last2;
if (n1 + 1 == n2)
return;
@@ -171,14 +161,14 @@ bessel_yn_r17 (gfc_array_r17 * const restrict ret, int n1, int n2,
#if defined(GFC_REAL_17_INFINITY)
if (unlikely (last2 == -GFC_REAL_17_INFINITY))
{
- ret->base_addr[i*stride] = -GFC_REAL_17_INFINITY;
+ GFC_DESCRIPTOR1_ELEM (ret, i) = -GFC_REAL_17_INFINITY;
}
else
#endif
{
- ret->base_addr[i*stride] = x2rev * (i-1+n1) * last2 - last1;
+ GFC_DESCRIPTOR1_ELEM (ret, i) = x2rev * (i-1+n1) * last2 - last1;
last1 = last2;
- last2 = ret->base_addr[i*stride];
+ last2 = GFC_DESCRIPTOR1_ELEM (ret, i);
}
}
}
diff --git a/libgfortran/generated/bessel_r4.c b/libgfortran/generated/bessel_r4.c
index 316609fbd512..7f746cc9c1ba 100644
--- a/libgfortran/generated/bessel_r4.c
+++ b/libgfortran/generated/bessel_r4.c
@@ -43,12 +43,9 @@ void
bessel_jn_r4 (gfc_array_r4 * const restrict ret, int n1, int n2, GFC_REAL_4 x)
{
int i;
- index_type stride;
GFC_REAL_4 last1, last2, x2rev;
- stride = GFC_DESCRIPTOR_STRIDE(ret,0);
-
if (ret->base_addr == NULL)
{
size_t size = n2 < n1 ? 0 : n2-n1+1;
@@ -66,24 +63,22 @@ bessel_jn_r4 (gfc_array_r4 * const restrict ret, int n1, int n2, GFC_REAL_4 x)
"(%ld vs. %ld)", (long int) n2-n1,
(long int) GFC_DESCRIPTOR_EXTENT(ret,0));
- stride = GFC_DESCRIPTOR_STRIDE(ret,0);
-
if (unlikely (x == 0))
{
ret->base_addr[0] = 1;
for (i = 1; i <= n2-n1; i++)
- ret->base_addr[i*stride] = 0;
+ GFC_DESCRIPTOR1_ELEM (ret, i) = 0;
return;
}
last1 = MATHFUNC(jn) (n2, x);
- ret->base_addr[(n2-n1)*stride] = last1;
+ GFC_DESCRIPTOR1_ELEM (ret, n2-n1) = last1;
if (n1 == n2)
return;
last2 = MATHFUNC(jn) (n2 - 1, x);
- ret->base_addr[(n2-n1-1)*stride] = last2;
+ GFC_DESCRIPTOR1_ELEM (ret, n2-n1-1) = last2;
if (n1 + 1 == n2)
return;
@@ -92,9 +87,9 @@ bessel_jn_r4 (gfc_array_r4 * const restrict ret, int n1, int n2, GFC_REAL_4 x)
for (i = n2-n1-2; i >= 0; i--)
{
- ret->base_addr[i*stride] = x2rev * (i+1+n1) * last2 - last1;
+ GFC_DESCRIPTOR1_ELEM (ret, i) = x2rev * (i+1+n1) * last2 - last1;
last1 = last2;
- last2 = ret->base_addr[i*stride];
+ last2 = GFC_DESCRIPTOR1_ELEM (ret, i);
}
}
@@ -110,12 +105,9 @@ bessel_yn_r4 (gfc_array_r4 * const restrict ret, int n1, int n2,
GFC_REAL_4 x)
{
int i;
- index_type stride;
GFC_REAL_4 last1, last2, x2rev;
- stride = GFC_DESCRIPTOR_STRIDE(ret,0);
-
if (ret->base_addr == NULL)
{
size_t size = n2 < n1 ? 0 : n2-n1+1;
@@ -133,27 +125,25 @@ bessel_yn_r4 (gfc_array_r4 * const restrict ret, int n1, int n2,
"(%ld vs. %ld)", (long int) n2-n1,
(long int) GFC_DESCRIPTOR_EXTENT(ret,0));
- stride = GFC_DESCRIPTOR_STRIDE(ret,0);
-
if (unlikely (x == 0))
{
for (i = 0; i <= n2-n1; i++)
#if defined(GFC_REAL_4_INFINITY)
- ret->base_addr[i*stride] = -GFC_REAL_4_INFINITY;
+ GFC_DESCRIPTOR1_ELEM (ret, i) = -GFC_REAL_4_INFINITY;
#else
- ret->base_addr[i*stride] = -GFC_REAL_4_HUGE;
+ GFC_DESCRIPTOR1_ELEM (ret, i) = -GFC_REAL_4_HUGE;
#endif
return;
}
last1 = MATHFUNC(yn) (n1, x);
- ret->base_addr[0] = last1;
+ GFC_DESCRIPTOR1_ELEM (ret, 0) = last1;
if (n1 == n2)
return;
last2 = MATHFUNC(yn) (n1 + 1, x);
- ret->base_addr[1*stride] = last2;
+ GFC_DESCRIPTOR1_ELEM (ret, 1) = last2;
if (n1 + 1 == n2)
return;
@@ -165,14 +155,14 @@ bessel_yn_r4 (gfc_array_r4 * const restrict ret, int n1, int n2,
#if defined(GFC_REAL_4_INFINITY)
if (unlikely (last2 == -GFC_REAL_4_INFINITY))
{
- ret->base_addr[i*stride] = -GFC_REAL_4_INFINITY;
+ GFC_DESCRIPTOR1_ELEM (ret, i) = -GFC_REAL_4_INFINITY;
}
else
#endif
{
- ret->base_addr[i*stride] = x2rev * (i-1+n1) * last2 - last1;
+ GFC_DESCRIPTOR1_ELEM (ret, i) = x2rev * (i-1+n1) * last2 - last1;
last1 = last2;
- last2 = ret->base_addr[i*stride];
+ last2 = GFC_DESCRIPTOR1_ELEM (ret, i);
}
}
}
diff --git a/libgfortran/generated/bessel_r8.c b/libgfortran/generated/bessel_r8.c
index 66878a21e8e6..ff90e439c0f1 100644
--- a/libgfortran/generated/bessel_r8.c
+++ b/libgfortran/generated/bessel_r8.c
@@ -43,12 +43,9 @@ void
bessel_jn_r8 (gfc_array_r8 * const restrict ret, int n1, int n2, GFC_REAL_8 x)
{
int i;
- index_type stride;
GFC_REAL_8 last1, last2, x2rev;
- stride = GFC_DESCRIPTOR_STRIDE(ret,0);
-
if (ret->base_addr == NULL)
{
size_t size = n2 < n1 ? 0 : n2-n1+1;
@@ -66,24 +63,22 @@ bessel_jn_r8 (gfc_array_r8 * const restrict ret, int n1, int n2, GFC_REAL_8 x)
"(%ld vs. %ld)", (long int) n2-n1,
(long int) GFC_DESCRIPTOR_EXTENT(ret,0));
- stride = GFC_DESCRIPTOR_STRIDE(ret,0);
-
if (unlikely (x == 0))
{
ret->base_addr[0] = 1;
for (i = 1; i <= n2-n1; i++)
- ret->base_addr[i*stride] = 0;
+ GFC_DESCRIPTOR1_ELEM (ret, i) = 0;
return;
}
last1 = MATHFUNC(jn) (n2, x);
- ret->base_addr[(n2-n1)*stride] = last1;
+ GFC_DESCRIPTOR1_ELEM (ret, n2-n1) = last1;
if (n1 == n2)
return;
last2 = MATHFUNC(jn) (n2 - 1, x);
- ret->base_addr[(n2-n1-1)*stride] = last2;
+ GFC_DESCRIPTOR1_ELEM (ret, n2-n1-1) = last2;
if (n1 + 1 == n2)
return;
@@ -92,9 +87,9 @@ bessel_jn_r8 (gfc_array_r8 * const restrict ret, int n1, int n2, GFC_REAL_8 x)
for (i = n2-n1-2; i >= 0; i--)
{
- ret->base_addr[i*stride] = x2rev * (i+1+n1) * last2 - last1;
+ GFC_DESCRIPTOR1_ELEM (ret, i) = x2rev * (i+1+n1) * last2 - last1;
last1 = last2;
- last2 = ret->base_addr[i*stride];
+ last2 = GFC_DESCRIPTOR1_ELEM (ret, i);
}
}
@@ -110,12 +105,9 @@ bessel_yn_r8 (gfc_array_r8 * const restrict ret, int n1, int n2,
GFC_REAL_8 x)
{
int i;
- index_type stride;
GFC_REAL_8 last1, last2, x2rev;
- stride = GFC_DESCRIPTOR_STRIDE(ret,0);
-
if (ret->base_addr == NULL)
{
size_t size = n2 < n1 ? 0 : n2-n1+1;
@@ -133,27 +125,25 @@ bessel_yn_r8 (gfc_array_r8 * const restrict ret, int n1, int n2,
"(%ld vs. %ld)", (long int) n2-n1,
(long int) GFC_DESCRIPTOR_EXTENT(ret,0));
- stride = GFC_DESCRIPTOR_STRIDE(ret,0);
-
if (unlikely (x == 0))
{
for (i = 0; i <= n2-n1; i++)
#if defined(GFC_REAL_8_INFINITY)
- ret->base_addr[i*stride] = -GFC_REAL_8_INFINITY;
+ GFC_DESCRIPTOR1_ELEM (ret, i) = -GFC_REAL_8_INFINITY;
#else
- ret->base_addr[i*stride] = -GFC_REAL_8_HUGE;
+ GFC_DESCRIPTOR1_ELEM (ret, i) = -GFC_REAL_8_HUGE;
#endif
return;
}
last1 = MATHFUNC(yn) (n1, x);
- ret->base_addr[0] = last1;
+ GFC_DESCRIPTOR1_ELEM (ret, 0) = last1;
if (n1 == n2)
return;
last2 = MATHFUNC(yn) (n1 + 1, x);
- ret->base_addr[1*stride] = last2;
+ GFC_DESCRIPTOR1_ELEM (ret, 1) = last2;
if (n1 + 1 == n2)
return;
@@ -165,14 +155,14 @@ bessel_yn_r8 (gfc_array_r8 * const restrict ret, int n1, int n2,
#if defined(GFC_REAL_8_INFINITY)
if (unlikely (last2 == -GFC_REAL_8_INFINITY))
{
- ret->base_addr[i*stride] = -GFC_REAL_8_INFINITY;
+ GFC_DESCRIPTOR1_ELEM (ret, i) = -GFC_REAL_8_INFINITY;
}
else
#endif
{
- ret->base_addr[i*stride] = x2rev * (i-1+n1) * last2 - last1;
+ GFC_DESCRIPTOR1_ELEM (ret, i) = x2rev * (i-1+n1) * last2 - last1;
last1 = last2;
- last2 = ret->base_addr[i*stride];
+ last2 = GFC_DESCRIPTOR1_ELEM (ret, i);
}
}
}
diff --git a/libgfortran/generated/cshift0_c10.c b/libgfortran/generated/cshift0_c10.c
index f7e2947aaf9d..3908c5b693f3 100644
--- a/libgfortran/generated/cshift0_c10.c
+++ b/libgfortran/generated/cshift0_c10.c
@@ -190,7 +190,7 @@ cshift0_c10 (gfc_array_c10 *ret, const gfc_array_c10 *array, ptrdiff_t shift,
/* Otherwise, we will have to perform the copy one element at
a time. */
GFC_COMPLEX_10 *dest = rptr;
- const GFC_COMPLEX_10 *src = (const GFC_COMPLEX_10 *) (((char*)sptr) + shift * soffset);
+ const GFC_COMPLEX_10 *src = PTR_ADD_OFFSET (sptr, shift * soffset);
for (n = 0; n < len - shift; n++)
{
diff --git a/libgfortran/generated/cshift0_c16.c b/libgfortran/generated/cshift0_c16.c
index 5326c1fb2833..962c009f3568 100644
--- a/libgfortran/generated/cshift0_c16.c
+++ b/libgfortran/generated/cshift0_c16.c
@@ -190,7 +190,7 @@ cshift0_c16 (gfc_array_c16 *ret, const gfc_array_c16 *array, ptrdiff_t shift,
/* Otherwise, we will have to perform the copy one element at
a time. */
GFC_COMPLEX_16 *dest = rptr;
- const GFC_COMPLEX_16 *src = (const GFC_COMPLEX_16 *) (((char*)sptr) + shift * soffset);
+ const GFC_COMPLEX_16 *src = PTR_ADD_OFFSET (sptr, shift * soffset);
for (n = 0; n < len - shift; n++)
{
diff --git a/libgfortran/generated/cshift0_c17.c b/libgfortran/generated/cshift0_c17.c
index 3309c8dfe96b..ebe965efcb5d 100644
--- a/libgfortran/generated/cshift0_c17.c
+++ b/libgfortran/generated/cshift0_c17.c
@@ -190,7 +190,7 @@ cshift0_c17 (gfc_array_c17 *ret, const gfc_array_c17 *array, ptrdiff_t shift,
/* Otherwise, we will have to perform the copy one element at
a time. */
GFC_COMPLEX_17 *dest = rptr;
- const GFC_COMPLEX_17 *src = (const GFC_COMPLEX_17 *) (((char*)sptr) + shift * soffset);
+ const GFC_COMPLEX_17 *src = PTR_ADD_OFFSET (sptr, shift * soffset);
for (n = 0; n < len - shift; n++)
{
diff --git a/libgfortran/generated/cshift0_c4.c b/libgfortran/generated/cshift0_c4.c
index 64090375ba00..71758c0aed18 100644
--- a/libgfortran/generated/cshift0_c4.c
+++ b/libgfortran/generated/cshift0_c4.c
@@ -190,7 +190,7 @@ cshift0_c4 (gfc_array_c4 *ret, const gfc_array_c4 *array, ptrdiff_t shift,
/* Otherwise, we will have to perform the copy one element at
a time. */
GFC_COMPLEX_4 *dest = rptr;
- const GFC_COMPLEX_4 *src = (const GFC_COMPLEX_4 *) (((char*)sptr) + shift * soffset);
+ const GFC_COMPLEX_4 *src = PTR_ADD_OFFSET (sptr, shift * soffset);
for (n = 0; n < len - shift; n++)
{
diff --git a/libgfortran/generated/cshift0_c8.c b/libgfortran/generated/cshift0_c8.c
index 756c0bb2bea2..1b9bff6e7627 100644
--- a/libgfortran/generated/cshift0_c8.c
+++ b/libgfortran/generated/cshift0_c8.c
@@ -190,7 +190,7 @@ cshift0_c8 (gfc_array_c8 *ret, const gfc_array_c8 *array, ptrdiff_t shift,
/* Otherwise, we will have to perform the copy one element at
a time. */
GFC_COMPLEX_8 *dest = rptr;
- const GFC_COMPLEX_8 *src = (const GFC_COMPLEX_8 *) (((char*)sptr) + shift * soffset);
+ const GFC_COMPLEX_8 *src = PTR_ADD_OFFSET (sptr, shift * soffset);
for (n = 0; n < len - shift; n++)
{
diff --git a/libgfortran/generated/cshift0_i1.c b/libgfortran/generated/cshift0_i1.c
index e2bfebd2ece0..c34a72e8612f 100644
--- a/libgfortran/generated/cshift0_i1.c
+++ b/libgfortran/generated/cshift0_i1.c
@@ -190,7 +190,7 @@ cshift0_i1 (gfc_array_i1 *ret, const gfc_array_i1 *array, ptrdiff_t shift,
/* Otherwise, we will have to perform the copy one element at
a time. */
GFC_INTEGER_1 *dest = rptr;
- const GFC_INTEGER_1 *src = (const GFC_INTEGER_1 *) (((char*)sptr) + shift * soffset);
+ const GFC_INTEGER_1 *src = PTR_ADD_OFFSET (sptr, shift * soffset);
for (n = 0; n < len - shift; n++)
{
diff --git a/libgfortran/generated/cshift0_i16.c b/libgfortran/generated/cshift0_i16.c
index b34d12a0bef3..01b502f5c538 100644
--- a/libgfortran/generated/cshift0_i16.c
+++ b/libgfortran/generated/cshift0_i16.c
@@ -190,7 +190,7 @@ cshift0_i16 (gfc_array_i16 *ret, const gfc_array_i16 *array, ptrdiff_t shift,
/* Otherwise, we will have to perform the copy one element at
a time. */
GFC_INTEGER_16 *dest = rptr;
- const GFC_INTEGER_16 *src = (const GFC_INTEGER_16 *) (((char*)sptr) + shift * soffset);
+ const GFC_INTEGER_16 *src = PTR_ADD_OFFSET (sptr, shift * soffset);
for (n = 0; n < len - shift; n++)
{
diff --git a/libgfortran/generated/cshift0_i2.c b/libgfortran/generated/cshift0_i2.c
index 07b975a34236..743cb6ececda 100644
--- a/libgfortran/generated/cshift0_i2.c
+++ b/libgfortran/generated/cshift0_i2.c
@@ -190,7 +190,7 @@ cshift0_i2 (gfc_array_i2 *ret, const gfc_array_i2 *array, ptrdiff_t shift,
/* Otherwise, we will have to perform the copy one element at
a time. */
GFC_INTEGER_2 *dest = rptr;
- const GFC_INTEGER_2 *src = (const GFC_INTEGER_2 *) (((char*)sptr) + shift * soffset);
+ const GFC_INTEGER_2 *src = PTR_ADD_OFFSET (sptr, shift * soffset);
for (n = 0; n < len - shift; n++)
{
diff --git a/libgfortran/generated/cshift0_i4.c b/libgfortran/generated/cshift0_i4.c
index 28ca14c35f3b..ccd1d2424a25 100644
--- a/libgfortran/generated/cshift0_i4.c
+++ b/libgfortran/generated/cshift0_i4.c
@@ -190,7 +190,7 @@ cshift0_i4 (gfc_array_i4 *ret, const gfc_array_i4 *array, ptrdiff_t shift,
/* Otherwise, we will have to perform the copy one element at
a time. */
GFC_INTEGER_4 *dest = rptr;
- const GFC_INTEGER_4 *src = (const GFC_INTEGER_4 *) (((char*)sptr) + shift * soffset);
+ const GFC_INTEGER_4 *src = PTR_ADD_OFFSET (sptr, shift * soffset);
for (n = 0; n < len - shift; n++)
{
diff --git a/libgfortran/generated/cshift0_i8.c b/libgfortran/generated/cshift0_i8.c
index 76f6f5aac62b..defbb1149c08 100644
--- a/libgfortran/generated/cshift0_i8.c
+++ b/libgfortran/generated/cshift0_i8.c
@@ -190,7 +190,7 @@ cshift0_i8 (gfc_array_i8 *ret, const gfc_array_i8 *array, ptrdiff_t shift,
/* Otherwise, we will have to perform the copy one element at
a time. */
GFC_INTEGER_8 *dest = rptr;
- const GFC_INTEGER_8 *src = (const GFC_INTEGER_8 *) (((char*)sptr) + shift * soffset);
+ const GFC_INTEGER_8 *src = PTR_ADD_OFFSET (sptr, shift * soffset);
for (n = 0; n < len - shift; n++)
{
diff --git a/libgfortran/generated/cshift0_r10.c b/libgfortran/generated/cshift0_r10.c
index 7b4406d3bad8..f40e42627e9e 100644
--- a/libgfortran/generated/cshift0_r10.c
+++ b/libgfortran/generated/cshift0_r10.c
@@ -190,7 +190,7 @@ cshift0_r10 (gfc_array_r10 *ret, const gfc_array_r10 *array, ptrdiff_t shift,
/* Otherwise, we will have to perform the copy one element at
a time. */
GFC_REAL_10 *dest = rptr;
- const GFC_REAL_10 *src = (const GFC_REAL_10 *) (((char*)sptr) + shift * soffset);
+ const GFC_REAL_10 *src = PTR_ADD_OFFSET (sptr, shift * soffset);
for (n = 0; n < len - shift; n++)
{
diff --git a/libgfortran/generated/cshift0_r16.c b/libgfortran/generated/cshift0_r16.c
index fea3900f3368..cbd98f3e44fd 100644
--- a/libgfortran/generated/cshift0_r16.c
+++ b/libgfortran/generated/cshift0_r16.c
@@ -190,7 +190,7 @@ cshift0_r16 (gfc_array_r16 *ret, const gfc_array_r16 *array, ptrdiff_t shift,
/* Otherwise, we will have to perform the copy one element at
a time. */
GFC_REAL_16 *dest = rptr;
- const GFC_REAL_16 *src = (const GFC_REAL_16 *) (((char*)sptr) + shift * soffset);
+ const GFC_REAL_16 *src = PTR_ADD_OFFSET (sptr, shift * soffset);
for (n = 0; n < len - shift; n++)
{
diff --git a/libgfortran/generated/cshift0_r17.c b/libgfortran/generated/cshift0_r17.c
index e4a056d7d146..a854dffc68c7 100644
--- a/libgfortran/generated/cshift0_r17.c
+++ b/libgfortran/generated/cshift0_r17.c
@@ -190,7 +190,7 @@ cshift0_r17 (gfc_array_r17 *ret, const gfc_array_r17 *array, ptrdiff_t shift,
/* Otherwise, we will have to perform the copy one element at
a time. */
GFC_REAL_17 *dest = rptr;
- const GFC_REAL_17 *src = (const GFC_REAL_17 *) (((char*)sptr) + shift * soffset);
+ const GFC_REAL_17 *src = PTR_ADD_OFFSET (sptr, shift * soffset);
for (n = 0; n < len - shift; n++)
{
diff --git a/libgfortran/generated/cshift0_r4.c b/libgfortran/generated/cshift0_r4.c
index 82f8182a17fa..cb4b93062dad 100644
--- a/libgfortran/generated/cshift0_r4.c
+++ b/libgfortran/generated/cshift0_r4.c
@@ -190,7 +190,7 @@ cshift0_r4 (gfc_array_r4 *ret, const gfc_array_r4 *array, ptrdiff_t shift,
/* Otherwise, we will have to perform the copy one element at
a time. */
GFC_REAL_4 *dest = rptr;
- const GFC_REAL_4 *src = (const GFC_REAL_4 *) (((char*)sptr) + shift * soffset);
+ const GFC_REAL_4 *src = PTR_ADD_OFFSET (sptr, shift * soffset);
for (n = 0; n < len - shift; n++)
{
diff --git a/libgfortran/generated/cshift0_r8.c b/libgfortran/generated/cshift0_r8.c
index 08da8bb81c73..8f5a67bffcc6 100644
--- a/libgfortran/generated/cshift0_r8.c
+++ b/libgfortran/generated/cshift0_r8.c
@@ -190,7 +190,7 @@ cshift0_r8 (gfc_array_r8 *ret, const gfc_array_r8 *array, ptrdiff_t shift,
/* Otherwise, we will have to perform the copy one element at
a time. */
GFC_REAL_8 *dest = rptr;
- const GFC_REAL_8 *src = (const GFC_REAL_8 *) (((char*)sptr) + shift * soffset);
+ const GFC_REAL_8 *src = PTR_ADD_OFFSET (sptr, shift * soffset);
for (n = 0; n < len - shift; n++)
{
diff --git a/libgfortran/generated/cshift1_16.c b/libgfortran/generated/cshift1_16.c
index 80e1a9ccb5a0..d7fc6012339e 100644
--- a/libgfortran/generated/cshift1_16.c
+++ b/libgfortran/generated/cshift1_16.c
@@ -263,7 +263,7 @@ cshift1 (gfc_array_char * const restrict ret,
sh += len;
}
- src = &sptr[sh * soffset];
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == size && roffset == size)
{
diff --git a/libgfortran/generated/cshift1_16_c10.c b/libgfortran/generated/cshift1_16_c10.c
index 82d97fda8317..d77f41b29ad8 100644
--- a/libgfortran/generated/cshift1_16_c10.c
+++ b/libgfortran/generated/cshift1_16_c10.c
@@ -133,7 +133,7 @@ cshift1_16_c10 (gfc_array_c10 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_COMPLEX_10 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_COMPLEX_10) && roffset == sizeof (GFC_COMPLEX_10))
{
diff --git a/libgfortran/generated/cshift1_16_c16.c b/libgfortran/generated/cshift1_16_c16.c
index 2f4c6b47d882..bbbb8534e2cd 100644
--- a/libgfortran/generated/cshift1_16_c16.c
+++ b/libgfortran/generated/cshift1_16_c16.c
@@ -133,7 +133,7 @@ cshift1_16_c16 (gfc_array_c16 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_COMPLEX_16 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_COMPLEX_16) && roffset == sizeof (GFC_COMPLEX_16))
{
diff --git a/libgfortran/generated/cshift1_16_c17.c b/libgfortran/generated/cshift1_16_c17.c
index 154021e50ee1..97473a49a4e9 100644
--- a/libgfortran/generated/cshift1_16_c17.c
+++ b/libgfortran/generated/cshift1_16_c17.c
@@ -133,7 +133,7 @@ cshift1_16_c17 (gfc_array_c17 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_COMPLEX_17 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_COMPLEX_17) && roffset == sizeof (GFC_COMPLEX_17))
{
diff --git a/libgfortran/generated/cshift1_16_c4.c b/libgfortran/generated/cshift1_16_c4.c
index a14bf888bc04..a4016c2fa6f4 100644
--- a/libgfortran/generated/cshift1_16_c4.c
+++ b/libgfortran/generated/cshift1_16_c4.c
@@ -133,7 +133,7 @@ cshift1_16_c4 (gfc_array_c4 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_COMPLEX_4 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_COMPLEX_4) && roffset == sizeof (GFC_COMPLEX_4))
{
diff --git a/libgfortran/generated/cshift1_16_c8.c b/libgfortran/generated/cshift1_16_c8.c
index 6ae7f93ec3a4..c80ec8a44ffe 100644
--- a/libgfortran/generated/cshift1_16_c8.c
+++ b/libgfortran/generated/cshift1_16_c8.c
@@ -133,7 +133,7 @@ cshift1_16_c8 (gfc_array_c8 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_COMPLEX_8 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_COMPLEX_8) && roffset == sizeof (GFC_COMPLEX_8))
{
diff --git a/libgfortran/generated/cshift1_16_i1.c b/libgfortran/generated/cshift1_16_i1.c
index f1027f85028d..c245f56ac116 100644
--- a/libgfortran/generated/cshift1_16_i1.c
+++ b/libgfortran/generated/cshift1_16_i1.c
@@ -133,7 +133,7 @@ cshift1_16_i1 (gfc_array_i1 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_INTEGER_1 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_INTEGER_1) && roffset == sizeof (GFC_INTEGER_1))
{
diff --git a/libgfortran/generated/cshift1_16_i16.c b/libgfortran/generated/cshift1_16_i16.c
index 29b24d5ad8af..aa98c12b1df2 100644
--- a/libgfortran/generated/cshift1_16_i16.c
+++ b/libgfortran/generated/cshift1_16_i16.c
@@ -133,7 +133,7 @@ cshift1_16_i16 (gfc_array_i16 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_INTEGER_16 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_INTEGER_16) && roffset == sizeof (GFC_INTEGER_16))
{
diff --git a/libgfortran/generated/cshift1_16_i2.c b/libgfortran/generated/cshift1_16_i2.c
index 27f38b6428b1..2b7faeb2b22d 100644
--- a/libgfortran/generated/cshift1_16_i2.c
+++ b/libgfortran/generated/cshift1_16_i2.c
@@ -133,7 +133,7 @@ cshift1_16_i2 (gfc_array_i2 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_INTEGER_2 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_INTEGER_2) && roffset == sizeof (GFC_INTEGER_2))
{
diff --git a/libgfortran/generated/cshift1_16_i4.c b/libgfortran/generated/cshift1_16_i4.c
index 4c1026a5bdec..682217316680 100644
--- a/libgfortran/generated/cshift1_16_i4.c
+++ b/libgfortran/generated/cshift1_16_i4.c
@@ -133,7 +133,7 @@ cshift1_16_i4 (gfc_array_i4 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_INTEGER_4 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_INTEGER_4) && roffset == sizeof (GFC_INTEGER_4))
{
diff --git a/libgfortran/generated/cshift1_16_i8.c b/libgfortran/generated/cshift1_16_i8.c
index e887c3e6d64a..151776087f68 100644
--- a/libgfortran/generated/cshift1_16_i8.c
+++ b/libgfortran/generated/cshift1_16_i8.c
@@ -133,7 +133,7 @@ cshift1_16_i8 (gfc_array_i8 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_INTEGER_8 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_INTEGER_8) && roffset == sizeof (GFC_INTEGER_8))
{
diff --git a/libgfortran/generated/cshift1_16_r10.c b/libgfortran/generated/cshift1_16_r10.c
index ba4671d2522c..5fadbd9e56b8 100644
--- a/libgfortran/generated/cshift1_16_r10.c
+++ b/libgfortran/generated/cshift1_16_r10.c
@@ -133,7 +133,7 @@ cshift1_16_r10 (gfc_array_r10 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_REAL_10 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_REAL_10) && roffset == sizeof (GFC_REAL_10))
{
diff --git a/libgfortran/generated/cshift1_16_r16.c b/libgfortran/generated/cshift1_16_r16.c
index b37066485799..bddc6beeb9d3 100644
--- a/libgfortran/generated/cshift1_16_r16.c
+++ b/libgfortran/generated/cshift1_16_r16.c
@@ -133,7 +133,7 @@ cshift1_16_r16 (gfc_array_r16 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_REAL_16 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_REAL_16) && roffset == sizeof (GFC_REAL_16))
{
diff --git a/libgfortran/generated/cshift1_16_r17.c b/libgfortran/generated/cshift1_16_r17.c
index 0503312a6fac..20a2b6c5d5d8 100644
--- a/libgfortran/generated/cshift1_16_r17.c
+++ b/libgfortran/generated/cshift1_16_r17.c
@@ -133,7 +133,7 @@ cshift1_16_r17 (gfc_array_r17 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_REAL_17 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_REAL_17) && roffset == sizeof (GFC_REAL_17))
{
diff --git a/libgfortran/generated/cshift1_16_r4.c b/libgfortran/generated/cshift1_16_r4.c
index 8fa510861ca9..8b95aecda015 100644
--- a/libgfortran/generated/cshift1_16_r4.c
+++ b/libgfortran/generated/cshift1_16_r4.c
@@ -133,7 +133,7 @@ cshift1_16_r4 (gfc_array_r4 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_REAL_4 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_REAL_4) && roffset == sizeof (GFC_REAL_4))
{
diff --git a/libgfortran/generated/cshift1_16_r8.c b/libgfortran/generated/cshift1_16_r8.c
index d6382f0925eb..5ebf723ce54f 100644
--- a/libgfortran/generated/cshift1_16_r8.c
+++ b/libgfortran/generated/cshift1_16_r8.c
@@ -133,7 +133,7 @@ cshift1_16_r8 (gfc_array_r8 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_REAL_8 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_REAL_8) && roffset == sizeof (GFC_REAL_8))
{
diff --git a/libgfortran/generated/cshift1_4.c b/libgfortran/generated/cshift1_4.c
index 72bea1b83913..ec92dbc48cda 100644
--- a/libgfortran/generated/cshift1_4.c
+++ b/libgfortran/generated/cshift1_4.c
@@ -263,7 +263,7 @@ cshift1 (gfc_array_char * const restrict ret,
sh += len;
}
- src = &sptr[sh * soffset];
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == size && roffset == size)
{
diff --git a/libgfortran/generated/cshift1_4_c10.c b/libgfortran/generated/cshift1_4_c10.c
index c941af142198..84d5163ece3b 100644
--- a/libgfortran/generated/cshift1_4_c10.c
+++ b/libgfortran/generated/cshift1_4_c10.c
@@ -133,7 +133,7 @@ cshift1_4_c10 (gfc_array_c10 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_COMPLEX_10 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_COMPLEX_10) && roffset == sizeof (GFC_COMPLEX_10))
{
diff --git a/libgfortran/generated/cshift1_4_c16.c b/libgfortran/generated/cshift1_4_c16.c
index 498d632b4ee4..4b5f36b9dbef 100644
--- a/libgfortran/generated/cshift1_4_c16.c
+++ b/libgfortran/generated/cshift1_4_c16.c
@@ -133,7 +133,7 @@ cshift1_4_c16 (gfc_array_c16 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_COMPLEX_16 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_COMPLEX_16) && roffset == sizeof (GFC_COMPLEX_16))
{
diff --git a/libgfortran/generated/cshift1_4_c17.c b/libgfortran/generated/cshift1_4_c17.c
index ad132772e73e..311a4c615057 100644
--- a/libgfortran/generated/cshift1_4_c17.c
+++ b/libgfortran/generated/cshift1_4_c17.c
@@ -133,7 +133,7 @@ cshift1_4_c17 (gfc_array_c17 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_COMPLEX_17 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_COMPLEX_17) && roffset == sizeof (GFC_COMPLEX_17))
{
diff --git a/libgfortran/generated/cshift1_4_c4.c b/libgfortran/generated/cshift1_4_c4.c
index 595d6607288b..13d64a7ca90e 100644
--- a/libgfortran/generated/cshift1_4_c4.c
+++ b/libgfortran/generated/cshift1_4_c4.c
@@ -133,7 +133,7 @@ cshift1_4_c4 (gfc_array_c4 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_COMPLEX_4 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_COMPLEX_4) && roffset == sizeof (GFC_COMPLEX_4))
{
diff --git a/libgfortran/generated/cshift1_4_c8.c b/libgfortran/generated/cshift1_4_c8.c
index 09b33f219a77..aa2911aef027 100644
--- a/libgfortran/generated/cshift1_4_c8.c
+++ b/libgfortran/generated/cshift1_4_c8.c
@@ -133,7 +133,7 @@ cshift1_4_c8 (gfc_array_c8 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_COMPLEX_8 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_COMPLEX_8) && roffset == sizeof (GFC_COMPLEX_8))
{
diff --git a/libgfortran/generated/cshift1_4_i1.c b/libgfortran/generated/cshift1_4_i1.c
index cbba837c1bf0..a3925ad7b10f 100644
--- a/libgfortran/generated/cshift1_4_i1.c
+++ b/libgfortran/generated/cshift1_4_i1.c
@@ -133,7 +133,7 @@ cshift1_4_i1 (gfc_array_i1 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_INTEGER_1 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_INTEGER_1) && roffset == sizeof (GFC_INTEGER_1))
{
diff --git a/libgfortran/generated/cshift1_4_i16.c b/libgfortran/generated/cshift1_4_i16.c
index fd0ab4fd2bb7..a9b2006054fc 100644
--- a/libgfortran/generated/cshift1_4_i16.c
+++ b/libgfortran/generated/cshift1_4_i16.c
@@ -133,7 +133,7 @@ cshift1_4_i16 (gfc_array_i16 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_INTEGER_16 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_INTEGER_16) && roffset == sizeof (GFC_INTEGER_16))
{
diff --git a/libgfortran/generated/cshift1_4_i2.c b/libgfortran/generated/cshift1_4_i2.c
index 16e31cb72b70..42ab18268b40 100644
--- a/libgfortran/generated/cshift1_4_i2.c
+++ b/libgfortran/generated/cshift1_4_i2.c
@@ -133,7 +133,7 @@ cshift1_4_i2 (gfc_array_i2 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_INTEGER_2 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_INTEGER_2) && roffset == sizeof (GFC_INTEGER_2))
{
diff --git a/libgfortran/generated/cshift1_4_i4.c b/libgfortran/generated/cshift1_4_i4.c
index 9808c8c3e2fd..9eb391c3e16c 100644
--- a/libgfortran/generated/cshift1_4_i4.c
+++ b/libgfortran/generated/cshift1_4_i4.c
@@ -133,7 +133,7 @@ cshift1_4_i4 (gfc_array_i4 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_INTEGER_4 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_INTEGER_4) && roffset == sizeof (GFC_INTEGER_4))
{
diff --git a/libgfortran/generated/cshift1_4_i8.c b/libgfortran/generated/cshift1_4_i8.c
index 8d88732320ce..c4238ccb0f90 100644
--- a/libgfortran/generated/cshift1_4_i8.c
+++ b/libgfortran/generated/cshift1_4_i8.c
@@ -133,7 +133,7 @@ cshift1_4_i8 (gfc_array_i8 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_INTEGER_8 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_INTEGER_8) && roffset == sizeof (GFC_INTEGER_8))
{
diff --git a/libgfortran/generated/cshift1_4_r10.c b/libgfortran/generated/cshift1_4_r10.c
index f389be10b00d..bdd0db006313 100644
--- a/libgfortran/generated/cshift1_4_r10.c
+++ b/libgfortran/generated/cshift1_4_r10.c
@@ -133,7 +133,7 @@ cshift1_4_r10 (gfc_array_r10 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_REAL_10 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_REAL_10) && roffset == sizeof (GFC_REAL_10))
{
diff --git a/libgfortran/generated/cshift1_4_r16.c b/libgfortran/generated/cshift1_4_r16.c
index 2fa14a47e590..5b3a77da6580 100644
--- a/libgfortran/generated/cshift1_4_r16.c
+++ b/libgfortran/generated/cshift1_4_r16.c
@@ -133,7 +133,7 @@ cshift1_4_r16 (gfc_array_r16 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_REAL_16 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_REAL_16) && roffset == sizeof (GFC_REAL_16))
{
diff --git a/libgfortran/generated/cshift1_4_r17.c b/libgfortran/generated/cshift1_4_r17.c
index 1458611a7f47..b9f14b466c1c 100644
--- a/libgfortran/generated/cshift1_4_r17.c
+++ b/libgfortran/generated/cshift1_4_r17.c
@@ -133,7 +133,7 @@ cshift1_4_r17 (gfc_array_r17 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_REAL_17 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_REAL_17) && roffset == sizeof (GFC_REAL_17))
{
diff --git a/libgfortran/generated/cshift1_4_r4.c b/libgfortran/generated/cshift1_4_r4.c
index 6a1e307e5a79..5f5001f1327d 100644
--- a/libgfortran/generated/cshift1_4_r4.c
+++ b/libgfortran/generated/cshift1_4_r4.c
@@ -133,7 +133,7 @@ cshift1_4_r4 (gfc_array_r4 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_REAL_4 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_REAL_4) && roffset == sizeof (GFC_REAL_4))
{
diff --git a/libgfortran/generated/cshift1_4_r8.c b/libgfortran/generated/cshift1_4_r8.c
index b8e7fcf8df64..34680890a632 100644
--- a/libgfortran/generated/cshift1_4_r8.c
+++ b/libgfortran/generated/cshift1_4_r8.c
@@ -133,7 +133,7 @@ cshift1_4_r8 (gfc_array_r8 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_REAL_8 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_REAL_8) && roffset == sizeof (GFC_REAL_8))
{
diff --git a/libgfortran/generated/cshift1_8.c b/libgfortran/generated/cshift1_8.c
index c9e682afe42e..db8f0df9d19c 100644
--- a/libgfortran/generated/cshift1_8.c
+++ b/libgfortran/generated/cshift1_8.c
@@ -263,7 +263,7 @@ cshift1 (gfc_array_char * const restrict ret,
sh += len;
}
- src = &sptr[sh * soffset];
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == size && roffset == size)
{
diff --git a/libgfortran/generated/cshift1_8_c10.c b/libgfortran/generated/cshift1_8_c10.c
index 618d2c54b3f2..051f11962f35 100644
--- a/libgfortran/generated/cshift1_8_c10.c
+++ b/libgfortran/generated/cshift1_8_c10.c
@@ -133,7 +133,7 @@ cshift1_8_c10 (gfc_array_c10 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_COMPLEX_10 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_COMPLEX_10) && roffset == sizeof (GFC_COMPLEX_10))
{
diff --git a/libgfortran/generated/cshift1_8_c16.c b/libgfortran/generated/cshift1_8_c16.c
index 66b9848960d8..e4da26cf24d1 100644
--- a/libgfortran/generated/cshift1_8_c16.c
+++ b/libgfortran/generated/cshift1_8_c16.c
@@ -133,7 +133,7 @@ cshift1_8_c16 (gfc_array_c16 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_COMPLEX_16 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_COMPLEX_16) && roffset == sizeof (GFC_COMPLEX_16))
{
diff --git a/libgfortran/generated/cshift1_8_c17.c b/libgfortran/generated/cshift1_8_c17.c
index 21d3709427c9..40a04791ac7f 100644
--- a/libgfortran/generated/cshift1_8_c17.c
+++ b/libgfortran/generated/cshift1_8_c17.c
@@ -133,7 +133,7 @@ cshift1_8_c17 (gfc_array_c17 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_COMPLEX_17 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_COMPLEX_17) && roffset == sizeof (GFC_COMPLEX_17))
{
diff --git a/libgfortran/generated/cshift1_8_c4.c b/libgfortran/generated/cshift1_8_c4.c
index 0088bb7e966f..e74079f549ca 100644
--- a/libgfortran/generated/cshift1_8_c4.c
+++ b/libgfortran/generated/cshift1_8_c4.c
@@ -133,7 +133,7 @@ cshift1_8_c4 (gfc_array_c4 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_COMPLEX_4 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_COMPLEX_4) && roffset == sizeof (GFC_COMPLEX_4))
{
diff --git a/libgfortran/generated/cshift1_8_c8.c b/libgfortran/generated/cshift1_8_c8.c
index 01a135da4f8f..f9e38a1c807b 100644
--- a/libgfortran/generated/cshift1_8_c8.c
+++ b/libgfortran/generated/cshift1_8_c8.c
@@ -133,7 +133,7 @@ cshift1_8_c8 (gfc_array_c8 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_COMPLEX_8 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_COMPLEX_8) && roffset == sizeof (GFC_COMPLEX_8))
{
diff --git a/libgfortran/generated/cshift1_8_i1.c b/libgfortran/generated/cshift1_8_i1.c
index 4e306300402c..b2a0c5c436bd 100644
--- a/libgfortran/generated/cshift1_8_i1.c
+++ b/libgfortran/generated/cshift1_8_i1.c
@@ -133,7 +133,7 @@ cshift1_8_i1 (gfc_array_i1 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_INTEGER_1 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_INTEGER_1) && roffset == sizeof (GFC_INTEGER_1))
{
diff --git a/libgfortran/generated/cshift1_8_i16.c b/libgfortran/generated/cshift1_8_i16.c
index 7dbd34b2b182..beaa7f9cf71d 100644
--- a/libgfortran/generated/cshift1_8_i16.c
+++ b/libgfortran/generated/cshift1_8_i16.c
@@ -133,7 +133,7 @@ cshift1_8_i16 (gfc_array_i16 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_INTEGER_16 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_INTEGER_16) && roffset == sizeof (GFC_INTEGER_16))
{
diff --git a/libgfortran/generated/cshift1_8_i2.c b/libgfortran/generated/cshift1_8_i2.c
index 80eb2f58ecdb..80c19b9bd965 100644
--- a/libgfortran/generated/cshift1_8_i2.c
+++ b/libgfortran/generated/cshift1_8_i2.c
@@ -133,7 +133,7 @@ cshift1_8_i2 (gfc_array_i2 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_INTEGER_2 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_INTEGER_2) && roffset == sizeof (GFC_INTEGER_2))
{
diff --git a/libgfortran/generated/cshift1_8_i4.c b/libgfortran/generated/cshift1_8_i4.c
index 48c4954abc56..fa7a43e18946 100644
--- a/libgfortran/generated/cshift1_8_i4.c
+++ b/libgfortran/generated/cshift1_8_i4.c
@@ -133,7 +133,7 @@ cshift1_8_i4 (gfc_array_i4 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_INTEGER_4 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_INTEGER_4) && roffset == sizeof (GFC_INTEGER_4))
{
diff --git a/libgfortran/generated/cshift1_8_i8.c b/libgfortran/generated/cshift1_8_i8.c
index 16605311c5d4..db5dadae0f2e 100644
--- a/libgfortran/generated/cshift1_8_i8.c
+++ b/libgfortran/generated/cshift1_8_i8.c
@@ -133,7 +133,7 @@ cshift1_8_i8 (gfc_array_i8 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_INTEGER_8 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_INTEGER_8) && roffset == sizeof (GFC_INTEGER_8))
{
diff --git a/libgfortran/generated/cshift1_8_r10.c b/libgfortran/generated/cshift1_8_r10.c
index ae92bbf2a505..74f47dd05543 100644
--- a/libgfortran/generated/cshift1_8_r10.c
+++ b/libgfortran/generated/cshift1_8_r10.c
@@ -133,7 +133,7 @@ cshift1_8_r10 (gfc_array_r10 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_REAL_10 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_REAL_10) && roffset == sizeof (GFC_REAL_10))
{
diff --git a/libgfortran/generated/cshift1_8_r16.c b/libgfortran/generated/cshift1_8_r16.c
index 9bd82ecd220d..fdf7bbef1d7a 100644
--- a/libgfortran/generated/cshift1_8_r16.c
+++ b/libgfortran/generated/cshift1_8_r16.c
@@ -133,7 +133,7 @@ cshift1_8_r16 (gfc_array_r16 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_REAL_16 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_REAL_16) && roffset == sizeof (GFC_REAL_16))
{
diff --git a/libgfortran/generated/cshift1_8_r17.c b/libgfortran/generated/cshift1_8_r17.c
index 8014b4491d19..fd91cb998d00 100644
--- a/libgfortran/generated/cshift1_8_r17.c
+++ b/libgfortran/generated/cshift1_8_r17.c
@@ -133,7 +133,7 @@ cshift1_8_r17 (gfc_array_r17 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_REAL_17 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_REAL_17) && roffset == sizeof (GFC_REAL_17))
{
diff --git a/libgfortran/generated/cshift1_8_r4.c b/libgfortran/generated/cshift1_8_r4.c
index b62263c7f227..98f2277d5f61 100644
--- a/libgfortran/generated/cshift1_8_r4.c
+++ b/libgfortran/generated/cshift1_8_r4.c
@@ -133,7 +133,7 @@ cshift1_8_r4 (gfc_array_r4 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_REAL_4 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_REAL_4) && roffset == sizeof (GFC_REAL_4))
{
diff --git a/libgfortran/generated/cshift1_8_r8.c b/libgfortran/generated/cshift1_8_r8.c
index 99d176fc8032..f7b5ecd119ca 100644
--- a/libgfortran/generated/cshift1_8_r8.c
+++ b/libgfortran/generated/cshift1_8_r8.c
@@ -133,7 +133,7 @@ cshift1_8_r8 (gfc_array_r8 * const restrict ret,
if (sh < 0)
sh += len;
}
- src = (const GFC_REAL_8 *) (((char*)sptr) + sh * soffset);
+ src = PTR_ADD_OFFSET (sptr, sh * soffset);
dest = rptr;
if (soffset == sizeof (GFC_REAL_8) && roffset == sizeof (GFC_REAL_8))
{
diff --git a/libgfortran/generated/findloc0_c10.c b/libgfortran/generated/findloc0_c10.c
index 96098c9db9c4..30224d4b7662 100644
--- a/libgfortran/generated/findloc0_c10.c
+++ b/libgfortran/generated/findloc0_c10.c
@@ -41,9 +41,7 @@ findloc0_c10 (gfc_array_index_type * const restrict retarray,
index_type count[GFC_MAX_DIMENSIONS];
index_type extent[GFC_MAX_DIMENSIONS];
index_type sstride[GFC_MAX_DIMENSIONS];
- index_type dstride;
const GFC_COMPLEX_10 *base;
- index_type * restrict dest;
index_type rank;
index_type n;
index_type sz;
@@ -66,12 +64,9 @@ findloc0_c10 (gfc_array_index_type * const restrict retarray,
"FINDLOC");
}
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
-
/* Set the return value. */
for (n = 0; n < rank; n++)
- dest[n * dstride] = 0;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0;
sz = 1;
for (n = 0; n < rank; n++)
@@ -88,7 +83,7 @@ findloc0_c10 (gfc_array_index_type * const restrict retarray,
if (back)
{
- base = array->base_addr + (sz - 1) * 1;
+ base = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, sz - 1);
while (1)
{
@@ -97,7 +92,7 @@ findloc0_c10 (gfc_array_index_type * const restrict retarray,
if (unlikely(*base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = extent[n] - count[n];
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = extent[n] - count[n];
return;
}
@@ -134,7 +129,7 @@ findloc0_c10 (gfc_array_index_type * const restrict retarray,
if (unlikely(*base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = count[n] + 1;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = count[n] + 1;
return;
}
@@ -178,9 +173,7 @@ mfindloc0_c10 (gfc_array_index_type * const restrict retarray,
index_type extent[GFC_MAX_DIMENSIONS];
index_type sstride[GFC_MAX_DIMENSIONS];
index_type mstride[GFC_MAX_DIMENSIONS];
- index_type dstride;
const GFC_COMPLEX_10 *base;
- index_type * restrict dest;
GFC_LOGICAL_1 *mbase;
index_type rank;
index_type n;
@@ -222,12 +215,9 @@ mfindloc0_c10 (gfc_array_index_type * const restrict retarray,
else
internal_error (NULL, "Funny sized logical array");
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
-
/* Set the return value. */
for (n = 0; n < rank; n++)
- dest[n * dstride] = 0;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0;
sz = 1;
for (n = 0; n < rank; n++)
@@ -245,7 +235,7 @@ mfindloc0_c10 (gfc_array_index_type * const restrict retarray,
if (back)
{
- base = array->base_addr + (sz - 1) * 1;
+ base = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, sz - 1);
mbase = mbase + (sz - 1) * mask_kind;
while (1)
{
@@ -254,7 +244,7 @@ mfindloc0_c10 (gfc_array_index_type * const restrict retarray,
if (unlikely(*mbase && *base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = extent[n] - count[n];
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = extent[n] - count[n];
return;
}
@@ -294,7 +284,7 @@ mfindloc0_c10 (gfc_array_index_type * const restrict retarray,
if (unlikely(*mbase && *base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = count[n] + 1;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = count[n] + 1;
return;
}
@@ -338,8 +328,6 @@ sfindloc0_c10 (gfc_array_index_type * const restrict retarray,
GFC_LOGICAL_4 * mask, GFC_LOGICAL_4 back)
{
index_type rank;
- index_type dstride;
- index_type * restrict dest;
index_type n;
if (mask == NULL || *mask)
@@ -366,10 +354,8 @@ sfindloc0_c10 (gfc_array_index_type * const restrict retarray,
"FINDLOC");
}
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
for (n = 0; n<rank; n++)
- dest[n * dstride] = 0 ;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0 ;
}
#endif
diff --git a/libgfortran/generated/findloc0_c16.c b/libgfortran/generated/findloc0_c16.c
index e36a31fc4e3c..425c6d53e8c7 100644
--- a/libgfortran/generated/findloc0_c16.c
+++ b/libgfortran/generated/findloc0_c16.c
@@ -41,9 +41,7 @@ findloc0_c16 (gfc_array_index_type * const restrict retarray,
index_type count[GFC_MAX_DIMENSIONS];
index_type extent[GFC_MAX_DIMENSIONS];
index_type sstride[GFC_MAX_DIMENSIONS];
- index_type dstride;
const GFC_COMPLEX_16 *base;
- index_type * restrict dest;
index_type rank;
index_type n;
index_type sz;
@@ -66,12 +64,9 @@ findloc0_c16 (gfc_array_index_type * const restrict retarray,
"FINDLOC");
}
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
-
/* Set the return value. */
for (n = 0; n < rank; n++)
- dest[n * dstride] = 0;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0;
sz = 1;
for (n = 0; n < rank; n++)
@@ -88,7 +83,7 @@ findloc0_c16 (gfc_array_index_type * const restrict retarray,
if (back)
{
- base = array->base_addr + (sz - 1) * 1;
+ base = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, sz - 1);
while (1)
{
@@ -97,7 +92,7 @@ findloc0_c16 (gfc_array_index_type * const restrict retarray,
if (unlikely(*base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = extent[n] - count[n];
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = extent[n] - count[n];
return;
}
@@ -134,7 +129,7 @@ findloc0_c16 (gfc_array_index_type * const restrict retarray,
if (unlikely(*base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = count[n] + 1;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = count[n] + 1;
return;
}
@@ -178,9 +173,7 @@ mfindloc0_c16 (gfc_array_index_type * const restrict retarray,
index_type extent[GFC_MAX_DIMENSIONS];
index_type sstride[GFC_MAX_DIMENSIONS];
index_type mstride[GFC_MAX_DIMENSIONS];
- index_type dstride;
const GFC_COMPLEX_16 *base;
- index_type * restrict dest;
GFC_LOGICAL_1 *mbase;
index_type rank;
index_type n;
@@ -222,12 +215,9 @@ mfindloc0_c16 (gfc_array_index_type * const restrict retarray,
else
internal_error (NULL, "Funny sized logical array");
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
-
/* Set the return value. */
for (n = 0; n < rank; n++)
- dest[n * dstride] = 0;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0;
sz = 1;
for (n = 0; n < rank; n++)
@@ -245,7 +235,7 @@ mfindloc0_c16 (gfc_array_index_type * const restrict retarray,
if (back)
{
- base = array->base_addr + (sz - 1) * 1;
+ base = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, sz - 1);
mbase = mbase + (sz - 1) * mask_kind;
while (1)
{
@@ -254,7 +244,7 @@ mfindloc0_c16 (gfc_array_index_type * const restrict retarray,
if (unlikely(*mbase && *base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = extent[n] - count[n];
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = extent[n] - count[n];
return;
}
@@ -294,7 +284,7 @@ mfindloc0_c16 (gfc_array_index_type * const restrict retarray,
if (unlikely(*mbase && *base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = count[n] + 1;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = count[n] + 1;
return;
}
@@ -338,8 +328,6 @@ sfindloc0_c16 (gfc_array_index_type * const restrict retarray,
GFC_LOGICAL_4 * mask, GFC_LOGICAL_4 back)
{
index_type rank;
- index_type dstride;
- index_type * restrict dest;
index_type n;
if (mask == NULL || *mask)
@@ -366,10 +354,8 @@ sfindloc0_c16 (gfc_array_index_type * const restrict retarray,
"FINDLOC");
}
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
for (n = 0; n<rank; n++)
- dest[n * dstride] = 0 ;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0 ;
}
#endif
diff --git a/libgfortran/generated/findloc0_c17.c b/libgfortran/generated/findloc0_c17.c
index bc5fc511d3a4..9b22a091fc8f 100644
--- a/libgfortran/generated/findloc0_c17.c
+++ b/libgfortran/generated/findloc0_c17.c
@@ -41,9 +41,7 @@ findloc0_c17 (gfc_array_index_type * const restrict retarray,
index_type count[GFC_MAX_DIMENSIONS];
index_type extent[GFC_MAX_DIMENSIONS];
index_type sstride[GFC_MAX_DIMENSIONS];
- index_type dstride;
const GFC_COMPLEX_17 *base;
- index_type * restrict dest;
index_type rank;
index_type n;
index_type sz;
@@ -66,12 +64,9 @@ findloc0_c17 (gfc_array_index_type * const restrict retarray,
"FINDLOC");
}
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
-
/* Set the return value. */
for (n = 0; n < rank; n++)
- dest[n * dstride] = 0;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0;
sz = 1;
for (n = 0; n < rank; n++)
@@ -88,7 +83,7 @@ findloc0_c17 (gfc_array_index_type * const restrict retarray,
if (back)
{
- base = array->base_addr + (sz - 1) * 1;
+ base = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, sz - 1);
while (1)
{
@@ -97,7 +92,7 @@ findloc0_c17 (gfc_array_index_type * const restrict retarray,
if (unlikely(*base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = extent[n] - count[n];
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = extent[n] - count[n];
return;
}
@@ -134,7 +129,7 @@ findloc0_c17 (gfc_array_index_type * const restrict retarray,
if (unlikely(*base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = count[n] + 1;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = count[n] + 1;
return;
}
@@ -178,9 +173,7 @@ mfindloc0_c17 (gfc_array_index_type * const restrict retarray,
index_type extent[GFC_MAX_DIMENSIONS];
index_type sstride[GFC_MAX_DIMENSIONS];
index_type mstride[GFC_MAX_DIMENSIONS];
- index_type dstride;
const GFC_COMPLEX_17 *base;
- index_type * restrict dest;
GFC_LOGICAL_1 *mbase;
index_type rank;
index_type n;
@@ -222,12 +215,9 @@ mfindloc0_c17 (gfc_array_index_type * const restrict retarray,
else
internal_error (NULL, "Funny sized logical array");
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
-
/* Set the return value. */
for (n = 0; n < rank; n++)
- dest[n * dstride] = 0;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0;
sz = 1;
for (n = 0; n < rank; n++)
@@ -245,7 +235,7 @@ mfindloc0_c17 (gfc_array_index_type * const restrict retarray,
if (back)
{
- base = array->base_addr + (sz - 1) * 1;
+ base = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, sz - 1);
mbase = mbase + (sz - 1) * mask_kind;
while (1)
{
@@ -254,7 +244,7 @@ mfindloc0_c17 (gfc_array_index_type * const restrict retarray,
if (unlikely(*mbase && *base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = extent[n] - count[n];
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = extent[n] - count[n];
return;
}
@@ -294,7 +284,7 @@ mfindloc0_c17 (gfc_array_index_type * const restrict retarray,
if (unlikely(*mbase && *base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = count[n] + 1;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = count[n] + 1;
return;
}
@@ -338,8 +328,6 @@ sfindloc0_c17 (gfc_array_index_type * const restrict retarray,
GFC_LOGICAL_4 * mask, GFC_LOGICAL_4 back)
{
index_type rank;
- index_type dstride;
- index_type * restrict dest;
index_type n;
if (mask == NULL || *mask)
@@ -366,10 +354,8 @@ sfindloc0_c17 (gfc_array_index_type * const restrict retarray,
"FINDLOC");
}
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
for (n = 0; n<rank; n++)
- dest[n * dstride] = 0 ;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0 ;
}
#endif
diff --git a/libgfortran/generated/findloc0_c4.c b/libgfortran/generated/findloc0_c4.c
index 6a92135328e0..57f40827d7b2 100644
--- a/libgfortran/generated/findloc0_c4.c
+++ b/libgfortran/generated/findloc0_c4.c
@@ -41,9 +41,7 @@ findloc0_c4 (gfc_array_index_type * const restrict retarray,
index_type count[GFC_MAX_DIMENSIONS];
index_type extent[GFC_MAX_DIMENSIONS];
index_type sstride[GFC_MAX_DIMENSIONS];
- index_type dstride;
const GFC_COMPLEX_4 *base;
- index_type * restrict dest;
index_type rank;
index_type n;
index_type sz;
@@ -66,12 +64,9 @@ findloc0_c4 (gfc_array_index_type * const restrict retarray,
"FINDLOC");
}
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
-
/* Set the return value. */
for (n = 0; n < rank; n++)
- dest[n * dstride] = 0;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0;
sz = 1;
for (n = 0; n < rank; n++)
@@ -88,7 +83,7 @@ findloc0_c4 (gfc_array_index_type * const restrict retarray,
if (back)
{
- base = array->base_addr + (sz - 1) * 1;
+ base = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, sz - 1);
while (1)
{
@@ -97,7 +92,7 @@ findloc0_c4 (gfc_array_index_type * const restrict retarray,
if (unlikely(*base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = extent[n] - count[n];
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = extent[n] - count[n];
return;
}
@@ -134,7 +129,7 @@ findloc0_c4 (gfc_array_index_type * const restrict retarray,
if (unlikely(*base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = count[n] + 1;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = count[n] + 1;
return;
}
@@ -178,9 +173,7 @@ mfindloc0_c4 (gfc_array_index_type * const restrict retarray,
index_type extent[GFC_MAX_DIMENSIONS];
index_type sstride[GFC_MAX_DIMENSIONS];
index_type mstride[GFC_MAX_DIMENSIONS];
- index_type dstride;
const GFC_COMPLEX_4 *base;
- index_type * restrict dest;
GFC_LOGICAL_1 *mbase;
index_type rank;
index_type n;
@@ -222,12 +215,9 @@ mfindloc0_c4 (gfc_array_index_type * const restrict retarray,
else
internal_error (NULL, "Funny sized logical array");
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
-
/* Set the return value. */
for (n = 0; n < rank; n++)
- dest[n * dstride] = 0;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0;
sz = 1;
for (n = 0; n < rank; n++)
@@ -245,7 +235,7 @@ mfindloc0_c4 (gfc_array_index_type * const restrict retarray,
if (back)
{
- base = array->base_addr + (sz - 1) * 1;
+ base = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, sz - 1);
mbase = mbase + (sz - 1) * mask_kind;
while (1)
{
@@ -254,7 +244,7 @@ mfindloc0_c4 (gfc_array_index_type * const restrict retarray,
if (unlikely(*mbase && *base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = extent[n] - count[n];
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = extent[n] - count[n];
return;
}
@@ -294,7 +284,7 @@ mfindloc0_c4 (gfc_array_index_type * const restrict retarray,
if (unlikely(*mbase && *base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = count[n] + 1;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = count[n] + 1;
return;
}
@@ -338,8 +328,6 @@ sfindloc0_c4 (gfc_array_index_type * const restrict retarray,
GFC_LOGICAL_4 * mask, GFC_LOGICAL_4 back)
{
index_type rank;
- index_type dstride;
- index_type * restrict dest;
index_type n;
if (mask == NULL || *mask)
@@ -366,10 +354,8 @@ sfindloc0_c4 (gfc_array_index_type * const restrict retarray,
"FINDLOC");
}
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
for (n = 0; n<rank; n++)
- dest[n * dstride] = 0 ;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0 ;
}
#endif
diff --git a/libgfortran/generated/findloc0_c8.c b/libgfortran/generated/findloc0_c8.c
index 7fffac1b5ca5..7cbfbb826f65 100644
--- a/libgfortran/generated/findloc0_c8.c
+++ b/libgfortran/generated/findloc0_c8.c
@@ -41,9 +41,7 @@ findloc0_c8 (gfc_array_index_type * const restrict retarray,
index_type count[GFC_MAX_DIMENSIONS];
index_type extent[GFC_MAX_DIMENSIONS];
index_type sstride[GFC_MAX_DIMENSIONS];
- index_type dstride;
const GFC_COMPLEX_8 *base;
- index_type * restrict dest;
index_type rank;
index_type n;
index_type sz;
@@ -66,12 +64,9 @@ findloc0_c8 (gfc_array_index_type * const restrict retarray,
"FINDLOC");
}
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
-
/* Set the return value. */
for (n = 0; n < rank; n++)
- dest[n * dstride] = 0;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0;
sz = 1;
for (n = 0; n < rank; n++)
@@ -88,7 +83,7 @@ findloc0_c8 (gfc_array_index_type * const restrict retarray,
if (back)
{
- base = array->base_addr + (sz - 1) * 1;
+ base = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, sz - 1);
while (1)
{
@@ -97,7 +92,7 @@ findloc0_c8 (gfc_array_index_type * const restrict retarray,
if (unlikely(*base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = extent[n] - count[n];
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = extent[n] - count[n];
return;
}
@@ -134,7 +129,7 @@ findloc0_c8 (gfc_array_index_type * const restrict retarray,
if (unlikely(*base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = count[n] + 1;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = count[n] + 1;
return;
}
@@ -178,9 +173,7 @@ mfindloc0_c8 (gfc_array_index_type * const restrict retarray,
index_type extent[GFC_MAX_DIMENSIONS];
index_type sstride[GFC_MAX_DIMENSIONS];
index_type mstride[GFC_MAX_DIMENSIONS];
- index_type dstride;
const GFC_COMPLEX_8 *base;
- index_type * restrict dest;
GFC_LOGICAL_1 *mbase;
index_type rank;
index_type n;
@@ -222,12 +215,9 @@ mfindloc0_c8 (gfc_array_index_type * const restrict retarray,
else
internal_error (NULL, "Funny sized logical array");
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
-
/* Set the return value. */
for (n = 0; n < rank; n++)
- dest[n * dstride] = 0;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0;
sz = 1;
for (n = 0; n < rank; n++)
@@ -245,7 +235,7 @@ mfindloc0_c8 (gfc_array_index_type * const restrict retarray,
if (back)
{
- base = array->base_addr + (sz - 1) * 1;
+ base = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, sz - 1);
mbase = mbase + (sz - 1) * mask_kind;
while (1)
{
@@ -254,7 +244,7 @@ mfindloc0_c8 (gfc_array_index_type * const restrict retarray,
if (unlikely(*mbase && *base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = extent[n] - count[n];
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = extent[n] - count[n];
return;
}
@@ -294,7 +284,7 @@ mfindloc0_c8 (gfc_array_index_type * const restrict retarray,
if (unlikely(*mbase && *base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = count[n] + 1;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = count[n] + 1;
return;
}
@@ -338,8 +328,6 @@ sfindloc0_c8 (gfc_array_index_type * const restrict retarray,
GFC_LOGICAL_4 * mask, GFC_LOGICAL_4 back)
{
index_type rank;
- index_type dstride;
- index_type * restrict dest;
index_type n;
if (mask == NULL || *mask)
@@ -366,10 +354,8 @@ sfindloc0_c8 (gfc_array_index_type * const restrict retarray,
"FINDLOC");
}
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
for (n = 0; n<rank; n++)
- dest[n * dstride] = 0 ;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0 ;
}
#endif
diff --git a/libgfortran/generated/findloc0_i1.c b/libgfortran/generated/findloc0_i1.c
index bed07f91924b..a82b3dfaee47 100644
--- a/libgfortran/generated/findloc0_i1.c
+++ b/libgfortran/generated/findloc0_i1.c
@@ -41,9 +41,7 @@ findloc0_i1 (gfc_array_index_type * const restrict retarray,
index_type count[GFC_MAX_DIMENSIONS];
index_type extent[GFC_MAX_DIMENSIONS];
index_type sstride[GFC_MAX_DIMENSIONS];
- index_type dstride;
const GFC_INTEGER_1 *base;
- index_type * restrict dest;
index_type rank;
index_type n;
index_type sz;
@@ -66,12 +64,9 @@ findloc0_i1 (gfc_array_index_type * const restrict retarray,
"FINDLOC");
}
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
-
/* Set the return value. */
for (n = 0; n < rank; n++)
- dest[n * dstride] = 0;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0;
sz = 1;
for (n = 0; n < rank; n++)
@@ -88,7 +83,7 @@ findloc0_i1 (gfc_array_index_type * const restrict retarray,
if (back)
{
- base = array->base_addr + (sz - 1) * 1;
+ base = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, sz - 1);
while (1)
{
@@ -97,7 +92,7 @@ findloc0_i1 (gfc_array_index_type * const restrict retarray,
if (unlikely(*base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = extent[n] - count[n];
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = extent[n] - count[n];
return;
}
@@ -134,7 +129,7 @@ findloc0_i1 (gfc_array_index_type * const restrict retarray,
if (unlikely(*base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = count[n] + 1;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = count[n] + 1;
return;
}
@@ -178,9 +173,7 @@ mfindloc0_i1 (gfc_array_index_type * const restrict retarray,
index_type extent[GFC_MAX_DIMENSIONS];
index_type sstride[GFC_MAX_DIMENSIONS];
index_type mstride[GFC_MAX_DIMENSIONS];
- index_type dstride;
const GFC_INTEGER_1 *base;
- index_type * restrict dest;
GFC_LOGICAL_1 *mbase;
index_type rank;
index_type n;
@@ -222,12 +215,9 @@ mfindloc0_i1 (gfc_array_index_type * const restrict retarray,
else
internal_error (NULL, "Funny sized logical array");
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
-
/* Set the return value. */
for (n = 0; n < rank; n++)
- dest[n * dstride] = 0;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0;
sz = 1;
for (n = 0; n < rank; n++)
@@ -245,7 +235,7 @@ mfindloc0_i1 (gfc_array_index_type * const restrict retarray,
if (back)
{
- base = array->base_addr + (sz - 1) * 1;
+ base = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, sz - 1);
mbase = mbase + (sz - 1) * mask_kind;
while (1)
{
@@ -254,7 +244,7 @@ mfindloc0_i1 (gfc_array_index_type * const restrict retarray,
if (unlikely(*mbase && *base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = extent[n] - count[n];
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = extent[n] - count[n];
return;
}
@@ -294,7 +284,7 @@ mfindloc0_i1 (gfc_array_index_type * const restrict retarray,
if (unlikely(*mbase && *base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = count[n] + 1;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = count[n] + 1;
return;
}
@@ -338,8 +328,6 @@ sfindloc0_i1 (gfc_array_index_type * const restrict retarray,
GFC_LOGICAL_4 * mask, GFC_LOGICAL_4 back)
{
index_type rank;
- index_type dstride;
- index_type * restrict dest;
index_type n;
if (mask == NULL || *mask)
@@ -366,10 +354,8 @@ sfindloc0_i1 (gfc_array_index_type * const restrict retarray,
"FINDLOC");
}
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
for (n = 0; n<rank; n++)
- dest[n * dstride] = 0 ;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0 ;
}
#endif
diff --git a/libgfortran/generated/findloc0_i16.c b/libgfortran/generated/findloc0_i16.c
index 53d802c8eda0..1f5602d447d3 100644
--- a/libgfortran/generated/findloc0_i16.c
+++ b/libgfortran/generated/findloc0_i16.c
@@ -41,9 +41,7 @@ findloc0_i16 (gfc_array_index_type * const restrict retarray,
index_type count[GFC_MAX_DIMENSIONS];
index_type extent[GFC_MAX_DIMENSIONS];
index_type sstride[GFC_MAX_DIMENSIONS];
- index_type dstride;
const GFC_INTEGER_16 *base;
- index_type * restrict dest;
index_type rank;
index_type n;
index_type sz;
@@ -66,12 +64,9 @@ findloc0_i16 (gfc_array_index_type * const restrict retarray,
"FINDLOC");
}
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
-
/* Set the return value. */
for (n = 0; n < rank; n++)
- dest[n * dstride] = 0;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0;
sz = 1;
for (n = 0; n < rank; n++)
@@ -88,7 +83,7 @@ findloc0_i16 (gfc_array_index_type * const restrict retarray,
if (back)
{
- base = array->base_addr + (sz - 1) * 1;
+ base = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, sz - 1);
while (1)
{
@@ -97,7 +92,7 @@ findloc0_i16 (gfc_array_index_type * const restrict retarray,
if (unlikely(*base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = extent[n] - count[n];
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = extent[n] - count[n];
return;
}
@@ -134,7 +129,7 @@ findloc0_i16 (gfc_array_index_type * const restrict retarray,
if (unlikely(*base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = count[n] + 1;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = count[n] + 1;
return;
}
@@ -178,9 +173,7 @@ mfindloc0_i16 (gfc_array_index_type * const restrict retarray,
index_type extent[GFC_MAX_DIMENSIONS];
index_type sstride[GFC_MAX_DIMENSIONS];
index_type mstride[GFC_MAX_DIMENSIONS];
- index_type dstride;
const GFC_INTEGER_16 *base;
- index_type * restrict dest;
GFC_LOGICAL_1 *mbase;
index_type rank;
index_type n;
@@ -222,12 +215,9 @@ mfindloc0_i16 (gfc_array_index_type * const restrict retarray,
else
internal_error (NULL, "Funny sized logical array");
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
-
/* Set the return value. */
for (n = 0; n < rank; n++)
- dest[n * dstride] = 0;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0;
sz = 1;
for (n = 0; n < rank; n++)
@@ -245,7 +235,7 @@ mfindloc0_i16 (gfc_array_index_type * const restrict retarray,
if (back)
{
- base = array->base_addr + (sz - 1) * 1;
+ base = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, sz - 1);
mbase = mbase + (sz - 1) * mask_kind;
while (1)
{
@@ -254,7 +244,7 @@ mfindloc0_i16 (gfc_array_index_type * const restrict retarray,
if (unlikely(*mbase && *base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = extent[n] - count[n];
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = extent[n] - count[n];
return;
}
@@ -294,7 +284,7 @@ mfindloc0_i16 (gfc_array_index_type * const restrict retarray,
if (unlikely(*mbase && *base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = count[n] + 1;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = count[n] + 1;
return;
}
@@ -338,8 +328,6 @@ sfindloc0_i16 (gfc_array_index_type * const restrict retarray,
GFC_LOGICAL_4 * mask, GFC_LOGICAL_4 back)
{
index_type rank;
- index_type dstride;
- index_type * restrict dest;
index_type n;
if (mask == NULL || *mask)
@@ -366,10 +354,8 @@ sfindloc0_i16 (gfc_array_index_type * const restrict retarray,
"FINDLOC");
}
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
for (n = 0; n<rank; n++)
- dest[n * dstride] = 0 ;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0 ;
}
#endif
diff --git a/libgfortran/generated/findloc0_i2.c b/libgfortran/generated/findloc0_i2.c
index a7ee428ecfd0..878a53ebc5ae 100644
--- a/libgfortran/generated/findloc0_i2.c
+++ b/libgfortran/generated/findloc0_i2.c
@@ -41,9 +41,7 @@ findloc0_i2 (gfc_array_index_type * const restrict retarray,
index_type count[GFC_MAX_DIMENSIONS];
index_type extent[GFC_MAX_DIMENSIONS];
index_type sstride[GFC_MAX_DIMENSIONS];
- index_type dstride;
const GFC_INTEGER_2 *base;
- index_type * restrict dest;
index_type rank;
index_type n;
index_type sz;
@@ -66,12 +64,9 @@ findloc0_i2 (gfc_array_index_type * const restrict retarray,
"FINDLOC");
}
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
-
/* Set the return value. */
for (n = 0; n < rank; n++)
- dest[n * dstride] = 0;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0;
sz = 1;
for (n = 0; n < rank; n++)
@@ -88,7 +83,7 @@ findloc0_i2 (gfc_array_index_type * const restrict retarray,
if (back)
{
- base = array->base_addr + (sz - 1) * 1;
+ base = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, sz - 1);
while (1)
{
@@ -97,7 +92,7 @@ findloc0_i2 (gfc_array_index_type * const restrict retarray,
if (unlikely(*base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = extent[n] - count[n];
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = extent[n] - count[n];
return;
}
@@ -134,7 +129,7 @@ findloc0_i2 (gfc_array_index_type * const restrict retarray,
if (unlikely(*base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = count[n] + 1;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = count[n] + 1;
return;
}
@@ -178,9 +173,7 @@ mfindloc0_i2 (gfc_array_index_type * const restrict retarray,
index_type extent[GFC_MAX_DIMENSIONS];
index_type sstride[GFC_MAX_DIMENSIONS];
index_type mstride[GFC_MAX_DIMENSIONS];
- index_type dstride;
const GFC_INTEGER_2 *base;
- index_type * restrict dest;
GFC_LOGICAL_1 *mbase;
index_type rank;
index_type n;
@@ -222,12 +215,9 @@ mfindloc0_i2 (gfc_array_index_type * const restrict retarray,
else
internal_error (NULL, "Funny sized logical array");
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
-
/* Set the return value. */
for (n = 0; n < rank; n++)
- dest[n * dstride] = 0;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0;
sz = 1;
for (n = 0; n < rank; n++)
@@ -245,7 +235,7 @@ mfindloc0_i2 (gfc_array_index_type * const restrict retarray,
if (back)
{
- base = array->base_addr + (sz - 1) * 1;
+ base = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, sz - 1);
mbase = mbase + (sz - 1) * mask_kind;
while (1)
{
@@ -254,7 +244,7 @@ mfindloc0_i2 (gfc_array_index_type * const restrict retarray,
if (unlikely(*mbase && *base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = extent[n] - count[n];
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = extent[n] - count[n];
return;
}
@@ -294,7 +284,7 @@ mfindloc0_i2 (gfc_array_index_type * const restrict retarray,
if (unlikely(*mbase && *base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = count[n] + 1;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = count[n] + 1;
return;
}
@@ -338,8 +328,6 @@ sfindloc0_i2 (gfc_array_index_type * const restrict retarray,
GFC_LOGICAL_4 * mask, GFC_LOGICAL_4 back)
{
index_type rank;
- index_type dstride;
- index_type * restrict dest;
index_type n;
if (mask == NULL || *mask)
@@ -366,10 +354,8 @@ sfindloc0_i2 (gfc_array_index_type * const restrict retarray,
"FINDLOC");
}
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
for (n = 0; n<rank; n++)
- dest[n * dstride] = 0 ;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0 ;
}
#endif
diff --git a/libgfortran/generated/findloc0_i4.c b/libgfortran/generated/findloc0_i4.c
index 7d3fa831391c..4615edd87359 100644
--- a/libgfortran/generated/findloc0_i4.c
+++ b/libgfortran/generated/findloc0_i4.c
@@ -41,9 +41,7 @@ findloc0_i4 (gfc_array_index_type * const restrict retarray,
index_type count[GFC_MAX_DIMENSIONS];
index_type extent[GFC_MAX_DIMENSIONS];
index_type sstride[GFC_MAX_DIMENSIONS];
- index_type dstride;
const GFC_INTEGER_4 *base;
- index_type * restrict dest;
index_type rank;
index_type n;
index_type sz;
@@ -66,12 +64,9 @@ findloc0_i4 (gfc_array_index_type * const restrict retarray,
"FINDLOC");
}
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
-
/* Set the return value. */
for (n = 0; n < rank; n++)
- dest[n * dstride] = 0;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0;
sz = 1;
for (n = 0; n < rank; n++)
@@ -88,7 +83,7 @@ findloc0_i4 (gfc_array_index_type * const restrict retarray,
if (back)
{
- base = array->base_addr + (sz - 1) * 1;
+ base = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, sz - 1);
while (1)
{
@@ -97,7 +92,7 @@ findloc0_i4 (gfc_array_index_type * const restrict retarray,
if (unlikely(*base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = extent[n] - count[n];
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = extent[n] - count[n];
return;
}
@@ -134,7 +129,7 @@ findloc0_i4 (gfc_array_index_type * const restrict retarray,
if (unlikely(*base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = count[n] + 1;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = count[n] + 1;
return;
}
@@ -178,9 +173,7 @@ mfindloc0_i4 (gfc_array_index_type * const restrict retarray,
index_type extent[GFC_MAX_DIMENSIONS];
index_type sstride[GFC_MAX_DIMENSIONS];
index_type mstride[GFC_MAX_DIMENSIONS];
- index_type dstride;
const GFC_INTEGER_4 *base;
- index_type * restrict dest;
GFC_LOGICAL_1 *mbase;
index_type rank;
index_type n;
@@ -222,12 +215,9 @@ mfindloc0_i4 (gfc_array_index_type * const restrict retarray,
else
internal_error (NULL, "Funny sized logical array");
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
-
/* Set the return value. */
for (n = 0; n < rank; n++)
- dest[n * dstride] = 0;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0;
sz = 1;
for (n = 0; n < rank; n++)
@@ -245,7 +235,7 @@ mfindloc0_i4 (gfc_array_index_type * const restrict retarray,
if (back)
{
- base = array->base_addr + (sz - 1) * 1;
+ base = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, sz - 1);
mbase = mbase + (sz - 1) * mask_kind;
while (1)
{
@@ -254,7 +244,7 @@ mfindloc0_i4 (gfc_array_index_type * const restrict retarray,
if (unlikely(*mbase && *base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = extent[n] - count[n];
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = extent[n] - count[n];
return;
}
@@ -294,7 +284,7 @@ mfindloc0_i4 (gfc_array_index_type * const restrict retarray,
if (unlikely(*mbase && *base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = count[n] + 1;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = count[n] + 1;
return;
}
@@ -338,8 +328,6 @@ sfindloc0_i4 (gfc_array_index_type * const restrict retarray,
GFC_LOGICAL_4 * mask, GFC_LOGICAL_4 back)
{
index_type rank;
- index_type dstride;
- index_type * restrict dest;
index_type n;
if (mask == NULL || *mask)
@@ -366,10 +354,8 @@ sfindloc0_i4 (gfc_array_index_type * const restrict retarray,
"FINDLOC");
}
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
for (n = 0; n<rank; n++)
- dest[n * dstride] = 0 ;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0 ;
}
#endif
diff --git a/libgfortran/generated/findloc0_i8.c b/libgfortran/generated/findloc0_i8.c
index a37bf091c43a..16f3a4ffba4c 100644
--- a/libgfortran/generated/findloc0_i8.c
+++ b/libgfortran/generated/findloc0_i8.c
@@ -41,9 +41,7 @@ findloc0_i8 (gfc_array_index_type * const restrict retarray,
index_type count[GFC_MAX_DIMENSIONS];
index_type extent[GFC_MAX_DIMENSIONS];
index_type sstride[GFC_MAX_DIMENSIONS];
- index_type dstride;
const GFC_INTEGER_8 *base;
- index_type * restrict dest;
index_type rank;
index_type n;
index_type sz;
@@ -66,12 +64,9 @@ findloc0_i8 (gfc_array_index_type * const restrict retarray,
"FINDLOC");
}
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
-
/* Set the return value. */
for (n = 0; n < rank; n++)
- dest[n * dstride] = 0;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0;
sz = 1;
for (n = 0; n < rank; n++)
@@ -88,7 +83,7 @@ findloc0_i8 (gfc_array_index_type * const restrict retarray,
if (back)
{
- base = array->base_addr + (sz - 1) * 1;
+ base = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, sz - 1);
while (1)
{
@@ -97,7 +92,7 @@ findloc0_i8 (gfc_array_index_type * const restrict retarray,
if (unlikely(*base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = extent[n] - count[n];
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = extent[n] - count[n];
return;
}
@@ -134,7 +129,7 @@ findloc0_i8 (gfc_array_index_type * const restrict retarray,
if (unlikely(*base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = count[n] + 1;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = count[n] + 1;
return;
}
@@ -178,9 +173,7 @@ mfindloc0_i8 (gfc_array_index_type * const restrict retarray,
index_type extent[GFC_MAX_DIMENSIONS];
index_type sstride[GFC_MAX_DIMENSIONS];
index_type mstride[GFC_MAX_DIMENSIONS];
- index_type dstride;
const GFC_INTEGER_8 *base;
- index_type * restrict dest;
GFC_LOGICAL_1 *mbase;
index_type rank;
index_type n;
@@ -222,12 +215,9 @@ mfindloc0_i8 (gfc_array_index_type * const restrict retarray,
else
internal_error (NULL, "Funny sized logical array");
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
-
/* Set the return value. */
for (n = 0; n < rank; n++)
- dest[n * dstride] = 0;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0;
sz = 1;
for (n = 0; n < rank; n++)
@@ -245,7 +235,7 @@ mfindloc0_i8 (gfc_array_index_type * const restrict retarray,
if (back)
{
- base = array->base_addr + (sz - 1) * 1;
+ base = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, sz - 1);
mbase = mbase + (sz - 1) * mask_kind;
while (1)
{
@@ -254,7 +244,7 @@ mfindloc0_i8 (gfc_array_index_type * const restrict retarray,
if (unlikely(*mbase && *base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = extent[n] - count[n];
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = extent[n] - count[n];
return;
}
@@ -294,7 +284,7 @@ mfindloc0_i8 (gfc_array_index_type * const restrict retarray,
if (unlikely(*mbase && *base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = count[n] + 1;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = count[n] + 1;
return;
}
@@ -338,8 +328,6 @@ sfindloc0_i8 (gfc_array_index_type * const restrict retarray,
GFC_LOGICAL_4 * mask, GFC_LOGICAL_4 back)
{
index_type rank;
- index_type dstride;
- index_type * restrict dest;
index_type n;
if (mask == NULL || *mask)
@@ -366,10 +354,8 @@ sfindloc0_i8 (gfc_array_index_type * const restrict retarray,
"FINDLOC");
}
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
for (n = 0; n<rank; n++)
- dest[n * dstride] = 0 ;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0 ;
}
#endif
diff --git a/libgfortran/generated/findloc0_r10.c b/libgfortran/generated/findloc0_r10.c
index 0a274e2cee64..7dd59b448a17 100644
--- a/libgfortran/generated/findloc0_r10.c
+++ b/libgfortran/generated/findloc0_r10.c
@@ -41,9 +41,7 @@ findloc0_r10 (gfc_array_index_type * const restrict retarray,
index_type count[GFC_MAX_DIMENSIONS];
index_type extent[GFC_MAX_DIMENSIONS];
index_type sstride[GFC_MAX_DIMENSIONS];
- index_type dstride;
const GFC_REAL_10 *base;
- index_type * restrict dest;
index_type rank;
index_type n;
index_type sz;
@@ -66,12 +64,9 @@ findloc0_r10 (gfc_array_index_type * const restrict retarray,
"FINDLOC");
}
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
-
/* Set the return value. */
for (n = 0; n < rank; n++)
- dest[n * dstride] = 0;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0;
sz = 1;
for (n = 0; n < rank; n++)
@@ -88,7 +83,7 @@ findloc0_r10 (gfc_array_index_type * const restrict retarray,
if (back)
{
- base = array->base_addr + (sz - 1) * 1;
+ base = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, sz - 1);
while (1)
{
@@ -97,7 +92,7 @@ findloc0_r10 (gfc_array_index_type * const restrict retarray,
if (unlikely(*base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = extent[n] - count[n];
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = extent[n] - count[n];
return;
}
@@ -134,7 +129,7 @@ findloc0_r10 (gfc_array_index_type * const restrict retarray,
if (unlikely(*base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = count[n] + 1;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = count[n] + 1;
return;
}
@@ -178,9 +173,7 @@ mfindloc0_r10 (gfc_array_index_type * const restrict retarray,
index_type extent[GFC_MAX_DIMENSIONS];
index_type sstride[GFC_MAX_DIMENSIONS];
index_type mstride[GFC_MAX_DIMENSIONS];
- index_type dstride;
const GFC_REAL_10 *base;
- index_type * restrict dest;
GFC_LOGICAL_1 *mbase;
index_type rank;
index_type n;
@@ -222,12 +215,9 @@ mfindloc0_r10 (gfc_array_index_type * const restrict retarray,
else
internal_error (NULL, "Funny sized logical array");
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
-
/* Set the return value. */
for (n = 0; n < rank; n++)
- dest[n * dstride] = 0;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0;
sz = 1;
for (n = 0; n < rank; n++)
@@ -245,7 +235,7 @@ mfindloc0_r10 (gfc_array_index_type * const restrict retarray,
if (back)
{
- base = array->base_addr + (sz - 1) * 1;
+ base = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, sz - 1);
mbase = mbase + (sz - 1) * mask_kind;
while (1)
{
@@ -254,7 +244,7 @@ mfindloc0_r10 (gfc_array_index_type * const restrict retarray,
if (unlikely(*mbase && *base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = extent[n] - count[n];
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = extent[n] - count[n];
return;
}
@@ -294,7 +284,7 @@ mfindloc0_r10 (gfc_array_index_type * const restrict retarray,
if (unlikely(*mbase && *base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = count[n] + 1;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = count[n] + 1;
return;
}
@@ -338,8 +328,6 @@ sfindloc0_r10 (gfc_array_index_type * const restrict retarray,
GFC_LOGICAL_4 * mask, GFC_LOGICAL_4 back)
{
index_type rank;
- index_type dstride;
- index_type * restrict dest;
index_type n;
if (mask == NULL || *mask)
@@ -366,10 +354,8 @@ sfindloc0_r10 (gfc_array_index_type * const restrict retarray,
"FINDLOC");
}
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
for (n = 0; n<rank; n++)
- dest[n * dstride] = 0 ;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0 ;
}
#endif
diff --git a/libgfortran/generated/findloc0_r16.c b/libgfortran/generated/findloc0_r16.c
index e3ca7605db9a..d4a24f5ca749 100644
--- a/libgfortran/generated/findloc0_r16.c
+++ b/libgfortran/generated/findloc0_r16.c
@@ -41,9 +41,7 @@ findloc0_r16 (gfc_array_index_type * const restrict retarray,
index_type count[GFC_MAX_DIMENSIONS];
index_type extent[GFC_MAX_DIMENSIONS];
index_type sstride[GFC_MAX_DIMENSIONS];
- index_type dstride;
const GFC_REAL_16 *base;
- index_type * restrict dest;
index_type rank;
index_type n;
index_type sz;
@@ -66,12 +64,9 @@ findloc0_r16 (gfc_array_index_type * const restrict retarray,
"FINDLOC");
}
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
-
/* Set the return value. */
for (n = 0; n < rank; n++)
- dest[n * dstride] = 0;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0;
sz = 1;
for (n = 0; n < rank; n++)
@@ -88,7 +83,7 @@ findloc0_r16 (gfc_array_index_type * const restrict retarray,
if (back)
{
- base = array->base_addr + (sz - 1) * 1;
+ base = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, sz - 1);
while (1)
{
@@ -97,7 +92,7 @@ findloc0_r16 (gfc_array_index_type * const restrict retarray,
if (unlikely(*base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = extent[n] - count[n];
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = extent[n] - count[n];
return;
}
@@ -134,7 +129,7 @@ findloc0_r16 (gfc_array_index_type * const restrict retarray,
if (unlikely(*base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = count[n] + 1;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = count[n] + 1;
return;
}
@@ -178,9 +173,7 @@ mfindloc0_r16 (gfc_array_index_type * const restrict retarray,
index_type extent[GFC_MAX_DIMENSIONS];
index_type sstride[GFC_MAX_DIMENSIONS];
index_type mstride[GFC_MAX_DIMENSIONS];
- index_type dstride;
const GFC_REAL_16 *base;
- index_type * restrict dest;
GFC_LOGICAL_1 *mbase;
index_type rank;
index_type n;
@@ -222,12 +215,9 @@ mfindloc0_r16 (gfc_array_index_type * const restrict retarray,
else
internal_error (NULL, "Funny sized logical array");
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
-
/* Set the return value. */
for (n = 0; n < rank; n++)
- dest[n * dstride] = 0;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0;
sz = 1;
for (n = 0; n < rank; n++)
@@ -245,7 +235,7 @@ mfindloc0_r16 (gfc_array_index_type * const restrict retarray,
if (back)
{
- base = array->base_addr + (sz - 1) * 1;
+ base = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, sz - 1);
mbase = mbase + (sz - 1) * mask_kind;
while (1)
{
@@ -254,7 +244,7 @@ mfindloc0_r16 (gfc_array_index_type * const restrict retarray,
if (unlikely(*mbase && *base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = extent[n] - count[n];
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = extent[n] - count[n];
return;
}
@@ -294,7 +284,7 @@ mfindloc0_r16 (gfc_array_index_type * const restrict retarray,
if (unlikely(*mbase && *base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = count[n] + 1;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = count[n] + 1;
return;
}
@@ -338,8 +328,6 @@ sfindloc0_r16 (gfc_array_index_type * const restrict retarray,
GFC_LOGICAL_4 * mask, GFC_LOGICAL_4 back)
{
index_type rank;
- index_type dstride;
- index_type * restrict dest;
index_type n;
if (mask == NULL || *mask)
@@ -366,10 +354,8 @@ sfindloc0_r16 (gfc_array_index_type * const restrict retarray,
"FINDLOC");
}
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
for (n = 0; n<rank; n++)
- dest[n * dstride] = 0 ;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0 ;
}
#endif
diff --git a/libgfortran/generated/findloc0_r17.c b/libgfortran/generated/findloc0_r17.c
index ef1d8919426f..d12e01372406 100644
--- a/libgfortran/generated/findloc0_r17.c
+++ b/libgfortran/generated/findloc0_r17.c
@@ -41,9 +41,7 @@ findloc0_r17 (gfc_array_index_type * const restrict retarray,
index_type count[GFC_MAX_DIMENSIONS];
index_type extent[GFC_MAX_DIMENSIONS];
index_type sstride[GFC_MAX_DIMENSIONS];
- index_type dstride;
const GFC_REAL_17 *base;
- index_type * restrict dest;
index_type rank;
index_type n;
index_type sz;
@@ -66,12 +64,9 @@ findloc0_r17 (gfc_array_index_type * const restrict retarray,
"FINDLOC");
}
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
-
/* Set the return value. */
for (n = 0; n < rank; n++)
- dest[n * dstride] = 0;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0;
sz = 1;
for (n = 0; n < rank; n++)
@@ -88,7 +83,7 @@ findloc0_r17 (gfc_array_index_type * const restrict retarray,
if (back)
{
- base = array->base_addr + (sz - 1) * 1;
+ base = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, sz - 1);
while (1)
{
@@ -97,7 +92,7 @@ findloc0_r17 (gfc_array_index_type * const restrict retarray,
if (unlikely(*base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = extent[n] - count[n];
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = extent[n] - count[n];
return;
}
@@ -134,7 +129,7 @@ findloc0_r17 (gfc_array_index_type * const restrict retarray,
if (unlikely(*base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = count[n] + 1;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = count[n] + 1;
return;
}
@@ -178,9 +173,7 @@ mfindloc0_r17 (gfc_array_index_type * const restrict retarray,
index_type extent[GFC_MAX_DIMENSIONS];
index_type sstride[GFC_MAX_DIMENSIONS];
index_type mstride[GFC_MAX_DIMENSIONS];
- index_type dstride;
const GFC_REAL_17 *base;
- index_type * restrict dest;
GFC_LOGICAL_1 *mbase;
index_type rank;
index_type n;
@@ -222,12 +215,9 @@ mfindloc0_r17 (gfc_array_index_type * const restrict retarray,
else
internal_error (NULL, "Funny sized logical array");
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
-
/* Set the return value. */
for (n = 0; n < rank; n++)
- dest[n * dstride] = 0;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0;
sz = 1;
for (n = 0; n < rank; n++)
@@ -245,7 +235,7 @@ mfindloc0_r17 (gfc_array_index_type * const restrict retarray,
if (back)
{
- base = array->base_addr + (sz - 1) * 1;
+ base = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, sz - 1);
mbase = mbase + (sz - 1) * mask_kind;
while (1)
{
@@ -254,7 +244,7 @@ mfindloc0_r17 (gfc_array_index_type * const restrict retarray,
if (unlikely(*mbase && *base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = extent[n] - count[n];
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = extent[n] - count[n];
return;
}
@@ -294,7 +284,7 @@ mfindloc0_r17 (gfc_array_index_type * const restrict retarray,
if (unlikely(*mbase && *base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = count[n] + 1;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = count[n] + 1;
return;
}
@@ -338,8 +328,6 @@ sfindloc0_r17 (gfc_array_index_type * const restrict retarray,
GFC_LOGICAL_4 * mask, GFC_LOGICAL_4 back)
{
index_type rank;
- index_type dstride;
- index_type * restrict dest;
index_type n;
if (mask == NULL || *mask)
@@ -366,10 +354,8 @@ sfindloc0_r17 (gfc_array_index_type * const restrict retarray,
"FINDLOC");
}
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
for (n = 0; n<rank; n++)
- dest[n * dstride] = 0 ;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0 ;
}
#endif
diff --git a/libgfortran/generated/findloc0_r4.c b/libgfortran/generated/findloc0_r4.c
index db5a3ead6975..f4f8d3fe55a2 100644
--- a/libgfortran/generated/findloc0_r4.c
+++ b/libgfortran/generated/findloc0_r4.c
@@ -41,9 +41,7 @@ findloc0_r4 (gfc_array_index_type * const restrict retarray,
index_type count[GFC_MAX_DIMENSIONS];
index_type extent[GFC_MAX_DIMENSIONS];
index_type sstride[GFC_MAX_DIMENSIONS];
- index_type dstride;
const GFC_REAL_4 *base;
- index_type * restrict dest;
index_type rank;
index_type n;
index_type sz;
@@ -66,12 +64,9 @@ findloc0_r4 (gfc_array_index_type * const restrict retarray,
"FINDLOC");
}
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
-
/* Set the return value. */
for (n = 0; n < rank; n++)
- dest[n * dstride] = 0;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0;
sz = 1;
for (n = 0; n < rank; n++)
@@ -88,7 +83,7 @@ findloc0_r4 (gfc_array_index_type * const restrict retarray,
if (back)
{
- base = array->base_addr + (sz - 1) * 1;
+ base = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, sz - 1);
while (1)
{
@@ -97,7 +92,7 @@ findloc0_r4 (gfc_array_index_type * const restrict retarray,
if (unlikely(*base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = extent[n] - count[n];
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = extent[n] - count[n];
return;
}
@@ -134,7 +129,7 @@ findloc0_r4 (gfc_array_index_type * const restrict retarray,
if (unlikely(*base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = count[n] + 1;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = count[n] + 1;
return;
}
@@ -178,9 +173,7 @@ mfindloc0_r4 (gfc_array_index_type * const restrict retarray,
index_type extent[GFC_MAX_DIMENSIONS];
index_type sstride[GFC_MAX_DIMENSIONS];
index_type mstride[GFC_MAX_DIMENSIONS];
- index_type dstride;
const GFC_REAL_4 *base;
- index_type * restrict dest;
GFC_LOGICAL_1 *mbase;
index_type rank;
index_type n;
@@ -222,12 +215,9 @@ mfindloc0_r4 (gfc_array_index_type * const restrict retarray,
else
internal_error (NULL, "Funny sized logical array");
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
-
/* Set the return value. */
for (n = 0; n < rank; n++)
- dest[n * dstride] = 0;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0;
sz = 1;
for (n = 0; n < rank; n++)
@@ -245,7 +235,7 @@ mfindloc0_r4 (gfc_array_index_type * const restrict retarray,
if (back)
{
- base = array->base_addr + (sz - 1) * 1;
+ base = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, sz - 1);
mbase = mbase + (sz - 1) * mask_kind;
while (1)
{
@@ -254,7 +244,7 @@ mfindloc0_r4 (gfc_array_index_type * const restrict retarray,
if (unlikely(*mbase && *base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = extent[n] - count[n];
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = extent[n] - count[n];
return;
}
@@ -294,7 +284,7 @@ mfindloc0_r4 (gfc_array_index_type * const restrict retarray,
if (unlikely(*mbase && *base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = count[n] + 1;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = count[n] + 1;
return;
}
@@ -338,8 +328,6 @@ sfindloc0_r4 (gfc_array_index_type * const restrict retarray,
GFC_LOGICAL_4 * mask, GFC_LOGICAL_4 back)
{
index_type rank;
- index_type dstride;
- index_type * restrict dest;
index_type n;
if (mask == NULL || *mask)
@@ -366,10 +354,8 @@ sfindloc0_r4 (gfc_array_index_type * const restrict retarray,
"FINDLOC");
}
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
for (n = 0; n<rank; n++)
- dest[n * dstride] = 0 ;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0 ;
}
#endif
diff --git a/libgfortran/generated/findloc0_r8.c b/libgfortran/generated/findloc0_r8.c
index 2314eeb7359d..6e505cd46104 100644
--- a/libgfortran/generated/findloc0_r8.c
+++ b/libgfortran/generated/findloc0_r8.c
@@ -41,9 +41,7 @@ findloc0_r8 (gfc_array_index_type * const restrict retarray,
index_type count[GFC_MAX_DIMENSIONS];
index_type extent[GFC_MAX_DIMENSIONS];
index_type sstride[GFC_MAX_DIMENSIONS];
- index_type dstride;
const GFC_REAL_8 *base;
- index_type * restrict dest;
index_type rank;
index_type n;
index_type sz;
@@ -66,12 +64,9 @@ findloc0_r8 (gfc_array_index_type * const restrict retarray,
"FINDLOC");
}
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
-
/* Set the return value. */
for (n = 0; n < rank; n++)
- dest[n * dstride] = 0;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0;
sz = 1;
for (n = 0; n < rank; n++)
@@ -88,7 +83,7 @@ findloc0_r8 (gfc_array_index_type * const restrict retarray,
if (back)
{
- base = array->base_addr + (sz - 1) * 1;
+ base = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, sz - 1);
while (1)
{
@@ -97,7 +92,7 @@ findloc0_r8 (gfc_array_index_type * const restrict retarray,
if (unlikely(*base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = extent[n] - count[n];
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = extent[n] - count[n];
return;
}
@@ -134,7 +129,7 @@ findloc0_r8 (gfc_array_index_type * const restrict retarray,
if (unlikely(*base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = count[n] + 1;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = count[n] + 1;
return;
}
@@ -178,9 +173,7 @@ mfindloc0_r8 (gfc_array_index_type * const restrict retarray,
index_type extent[GFC_MAX_DIMENSIONS];
index_type sstride[GFC_MAX_DIMENSIONS];
index_type mstride[GFC_MAX_DIMENSIONS];
- index_type dstride;
const GFC_REAL_8 *base;
- index_type * restrict dest;
GFC_LOGICAL_1 *mbase;
index_type rank;
index_type n;
@@ -222,12 +215,9 @@ mfindloc0_r8 (gfc_array_index_type * const restrict retarray,
else
internal_error (NULL, "Funny sized logical array");
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
-
/* Set the return value. */
for (n = 0; n < rank; n++)
- dest[n * dstride] = 0;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0;
sz = 1;
for (n = 0; n < rank; n++)
@@ -245,7 +235,7 @@ mfindloc0_r8 (gfc_array_index_type * const restrict retarray,
if (back)
{
- base = array->base_addr + (sz - 1) * 1;
+ base = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, sz - 1);
mbase = mbase + (sz - 1) * mask_kind;
while (1)
{
@@ -254,7 +244,7 @@ mfindloc0_r8 (gfc_array_index_type * const restrict retarray,
if (unlikely(*mbase && *base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = extent[n] - count[n];
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = extent[n] - count[n];
return;
}
@@ -294,7 +284,7 @@ mfindloc0_r8 (gfc_array_index_type * const restrict retarray,
if (unlikely(*mbase && *base == value))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = count[n] + 1;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = count[n] + 1;
return;
}
@@ -338,8 +328,6 @@ sfindloc0_r8 (gfc_array_index_type * const restrict retarray,
GFC_LOGICAL_4 * mask, GFC_LOGICAL_4 back)
{
index_type rank;
- index_type dstride;
- index_type * restrict dest;
index_type n;
if (mask == NULL || *mask)
@@ -366,10 +354,8 @@ sfindloc0_r8 (gfc_array_index_type * const restrict retarray,
"FINDLOC");
}
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
for (n = 0; n<rank; n++)
- dest[n * dstride] = 0 ;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0 ;
}
#endif
diff --git a/libgfortran/generated/findloc0_s1.c b/libgfortran/generated/findloc0_s1.c
index ba72c39fc528..40d48d46ae8c 100644
--- a/libgfortran/generated/findloc0_s1.c
+++ b/libgfortran/generated/findloc0_s1.c
@@ -42,9 +42,7 @@ findloc0_s1 (gfc_array_index_type * const restrict retarray,
index_type count[GFC_MAX_DIMENSIONS];
index_type extent[GFC_MAX_DIMENSIONS];
index_type sstride[GFC_MAX_DIMENSIONS];
- index_type dstride;
const GFC_UINTEGER_1 *base;
- index_type * restrict dest;
index_type rank;
index_type n;
index_type sz;
@@ -67,12 +65,9 @@ findloc0_s1 (gfc_array_index_type * const restrict retarray,
"FINDLOC");
}
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
-
/* Set the return value. */
for (n = 0; n < rank; n++)
- dest[n * dstride] = 0;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0;
sz = 1;
for (n = 0; n < rank; n++)
@@ -89,7 +84,7 @@ findloc0_s1 (gfc_array_index_type * const restrict retarray,
if (back)
{
- base = array->base_addr + (sz - 1) * len_array;
+ base = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, sz - 1);
while (1)
{
@@ -98,7 +93,7 @@ findloc0_s1 (gfc_array_index_type * const restrict retarray,
if (unlikely(compare_string (len_array, (char *) base, len_value, (char *) value) == 0))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = extent[n] - count[n];
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = extent[n] - count[n];
return;
}
@@ -135,7 +130,7 @@ findloc0_s1 (gfc_array_index_type * const restrict retarray,
if (unlikely(compare_string (len_array, (char *) base, len_value, (char *) value) == 0))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = count[n] + 1;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = count[n] + 1;
return;
}
@@ -181,9 +176,7 @@ mfindloc0_s1 (gfc_array_index_type * const restrict retarray,
index_type extent[GFC_MAX_DIMENSIONS];
index_type sstride[GFC_MAX_DIMENSIONS];
index_type mstride[GFC_MAX_DIMENSIONS];
- index_type dstride;
const GFC_UINTEGER_1 *base;
- index_type * restrict dest;
GFC_LOGICAL_1 *mbase;
index_type rank;
index_type n;
@@ -225,12 +218,9 @@ mfindloc0_s1 (gfc_array_index_type * const restrict retarray,
else
internal_error (NULL, "Funny sized logical array");
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
-
/* Set the return value. */
for (n = 0; n < rank; n++)
- dest[n * dstride] = 0;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0;
sz = 1;
for (n = 0; n < rank; n++)
@@ -248,7 +238,7 @@ mfindloc0_s1 (gfc_array_index_type * const restrict retarray,
if (back)
{
- base = array->base_addr + (sz - 1) * len_array;
+ base = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, sz - 1);
mbase = mbase + (sz - 1) * mask_kind;
while (1)
{
@@ -257,7 +247,7 @@ mfindloc0_s1 (gfc_array_index_type * const restrict retarray,
if (unlikely(*mbase && compare_string (len_array, (char *) base, len_value, (char *) value) == 0))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = extent[n] - count[n];
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = extent[n] - count[n];
return;
}
@@ -297,7 +287,7 @@ mfindloc0_s1 (gfc_array_index_type * const restrict retarray,
if (unlikely(*mbase && compare_string (len_array, (char *) base, len_value, (char *) value) == 0))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = count[n] + 1;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = count[n] + 1;
return;
}
@@ -343,8 +333,6 @@ sfindloc0_s1 (gfc_array_index_type * const restrict retarray,
gfc_charlen_type len_value)
{
index_type rank;
- index_type dstride;
- index_type * restrict dest;
index_type n;
if (mask == NULL || *mask)
@@ -371,10 +359,8 @@ sfindloc0_s1 (gfc_array_index_type * const restrict retarray,
"FINDLOC");
}
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
for (n = 0; n<rank; n++)
- dest[n * dstride] = 0 ;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0 ;
}
#endif
diff --git a/libgfortran/generated/findloc0_s4.c b/libgfortran/generated/findloc0_s4.c
index b963db3fa16e..245aff165358 100644
--- a/libgfortran/generated/findloc0_s4.c
+++ b/libgfortran/generated/findloc0_s4.c
@@ -42,9 +42,7 @@ findloc0_s4 (gfc_array_index_type * const restrict retarray,
index_type count[GFC_MAX_DIMENSIONS];
index_type extent[GFC_MAX_DIMENSIONS];
index_type sstride[GFC_MAX_DIMENSIONS];
- index_type dstride;
const GFC_UINTEGER_4 *base;
- index_type * restrict dest;
index_type rank;
index_type n;
index_type sz;
@@ -67,12 +65,9 @@ findloc0_s4 (gfc_array_index_type * const restrict retarray,
"FINDLOC");
}
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
-
/* Set the return value. */
for (n = 0; n < rank; n++)
- dest[n * dstride] = 0;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0;
sz = 1;
for (n = 0; n < rank; n++)
@@ -89,7 +84,7 @@ findloc0_s4 (gfc_array_index_type * const restrict retarray,
if (back)
{
- base = array->base_addr + (sz - 1) * len_array;
+ base = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, sz - 1);
while (1)
{
@@ -98,7 +93,7 @@ findloc0_s4 (gfc_array_index_type * const restrict retarray,
if (unlikely(compare_string_char4 (len_array, base, len_value, value) == 0))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = extent[n] - count[n];
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = extent[n] - count[n];
return;
}
@@ -135,7 +130,7 @@ findloc0_s4 (gfc_array_index_type * const restrict retarray,
if (unlikely(compare_string_char4 (len_array, base, len_value, value) == 0))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = count[n] + 1;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = count[n] + 1;
return;
}
@@ -181,9 +176,7 @@ mfindloc0_s4 (gfc_array_index_type * const restrict retarray,
index_type extent[GFC_MAX_DIMENSIONS];
index_type sstride[GFC_MAX_DIMENSIONS];
index_type mstride[GFC_MAX_DIMENSIONS];
- index_type dstride;
const GFC_UINTEGER_4 *base;
- index_type * restrict dest;
GFC_LOGICAL_1 *mbase;
index_type rank;
index_type n;
@@ -225,12 +218,9 @@ mfindloc0_s4 (gfc_array_index_type * const restrict retarray,
else
internal_error (NULL, "Funny sized logical array");
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
-
/* Set the return value. */
for (n = 0; n < rank; n++)
- dest[n * dstride] = 0;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0;
sz = 1;
for (n = 0; n < rank; n++)
@@ -248,7 +238,7 @@ mfindloc0_s4 (gfc_array_index_type * const restrict retarray,
if (back)
{
- base = array->base_addr + (sz - 1) * len_array;
+ base = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, sz - 1);
mbase = mbase + (sz - 1) * mask_kind;
while (1)
{
@@ -257,7 +247,7 @@ mfindloc0_s4 (gfc_array_index_type * const restrict retarray,
if (unlikely(*mbase && compare_string_char4 (len_array, base, len_value, value) == 0))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = extent[n] - count[n];
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = extent[n] - count[n];
return;
}
@@ -297,7 +287,7 @@ mfindloc0_s4 (gfc_array_index_type * const restrict retarray,
if (unlikely(*mbase && compare_string_char4 (len_array, base, len_value, value) == 0))
{
for (n = 0; n < rank; n++)
- dest[n * dstride] = count[n] + 1;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = count[n] + 1;
return;
}
@@ -343,8 +333,6 @@ sfindloc0_s4 (gfc_array_index_type * const restrict retarray,
gfc_charlen_type len_value)
{
index_type rank;
- index_type dstride;
- index_type * restrict dest;
index_type n;
if (mask == NULL || *mask)
@@ -371,10 +359,8 @@ sfindloc0_s4 (gfc_array_index_type * const restrict retarray,
"FINDLOC");
}
- dstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- dest = retarray->base_addr;
for (n = 0; n<rank; n++)
- dest[n * dstride] = 0 ;
+ GFC_DESCRIPTOR1_ELEM (retarray, n) = 0 ;
}
#endif
diff --git a/libgfortran/generated/findloc1_c10.c b/libgfortran/generated/findloc1_c10.c
index 90bc1c5e86ec..cc099b689bd7 100644
--- a/libgfortran/generated/findloc1_c10.c
+++ b/libgfortran/generated/findloc1_c10.c
@@ -139,7 +139,7 @@ findloc1_c10 (gfc_array_index_type * const restrict retarray,
result = 0;
if (back)
{
- src = (const GFC_COMPLEX_10 * restrict) (((char*) base) + (len - 1) * delta);
+ src = PTR_ADD_OFFSET (base, (len - 1) * delta);
for (n = len; n > 0; n--)
{
if (*src == value)
@@ -325,7 +325,7 @@ mfindloc1_c10 (gfc_array_index_type * const restrict retarray,
result = 0;
if (back)
{
- src = (const GFC_COMPLEX_10 * restrict) (((char*)base) + (len - 1) * delta);
+ src = PTR_ADD_OFFSET (base, (len - 1) * delta);
msrc = mbase + (len - 1) * mdelta;
for (n = len; n > 0; n--)
{
diff --git a/libgfortran/generated/findloc1_c16.c b/libgfortran/generated/findloc1_c16.c
index 3823bbb7c025..83cee88cb95d 100644
--- a/libgfortran/generated/findloc1_c16.c
+++ b/libgfortran/generated/findloc1_c16.c
@@ -139,7 +139,7 @@ findloc1_c16 (gfc_array_index_type * const restrict retarray,
result = 0;
if (back)
{
- src = (const GFC_COMPLEX_16 * restrict) (((char*) base) + (len - 1) * delta);
+ src = PTR_ADD_OFFSET (base, (len - 1) * delta);
for (n = len; n > 0; n--)
{
if (*src == value)
@@ -325,7 +325,7 @@ mfindloc1_c16 (gfc_array_index_type * const restrict retarray,
result = 0;
if (back)
{
- src = (const GFC_COMPLEX_16 * restrict) (((char*)base) + (len - 1) * delta);
+ src = PTR_ADD_OFFSET (base, (len - 1) * delta);
msrc = mbase + (len - 1) * mdelta;
for (n = len; n > 0; n--)
{
diff --git a/libgfortran/generated/findloc1_c17.c b/libgfortran/generated/findloc1_c17.c
index 6039412b1d1f..c6392edcce34 100644
--- a/libgfortran/generated/findloc1_c17.c
+++ b/libgfortran/generated/findloc1_c17.c
@@ -139,7 +139,7 @@ findloc1_c17 (gfc_array_index_type * const restrict retarray,
result = 0;
if (back)
{
- src = (const GFC_COMPLEX_17 * restrict) (((char*) base) + (len - 1) * delta);
+ src = PTR_ADD_OFFSET (base, (len - 1) * delta);
for (n = len; n > 0; n--)
{
if (*src == value)
@@ -325,7 +325,7 @@ mfindloc1_c17 (gfc_array_index_type * const restrict retarray,
result = 0;
if (back)
{
- src = (const GFC_COMPLEX_17 * restrict) (((char*)base) + (len - 1) * delta);
+ src = PTR_ADD_OFFSET (base, (len - 1) * delta);
msrc = mbase + (len - 1) * mdelta;
for (n = len; n > 0; n--)
{
diff --git a/libgfortran/generated/findloc1_c4.c b/libgfortran/generated/findloc1_c4.c
index 8907721eb429..be8551fc29aa 100644
--- a/libgfortran/generated/findloc1_c4.c
+++ b/libgfortran/generated/findloc1_c4.c
@@ -139,7 +139,7 @@ findloc1_c4 (gfc_array_index_type * const restrict retarray,
result = 0;
if (back)
{
- src = (const GFC_COMPLEX_4 * restrict) (((char*) base) + (len - 1) * delta);
+ src = PTR_ADD_OFFSET (base, (len - 1) * delta);
for (n = len; n > 0; n--)
{
if (*src == value)
@@ -325,7 +325,7 @@ mfindloc1_c4 (gfc_array_index_type * const restrict retarray,
result = 0;
if (back)
{
- src = (const GFC_COMPLEX_4 * restrict) (((char*)base) + (len - 1) * delta);
+ src = PTR_ADD_OFFSET (base, (len - 1) * delta);
msrc = mbase + (len - 1) * mdelta;
for (n = len; n > 0; n--)
{
diff --git a/libgfortran/generated/findloc1_c8.c b/libgfortran/generated/findloc1_c8.c
index 8d0e2ac568e5..1bbef53695fa 100644
--- a/libgfortran/generated/findloc1_c8.c
+++ b/libgfortran/generated/findloc1_c8.c
@@ -139,7 +139,7 @@ findloc1_c8 (gfc_array_index_type * const restrict retarray,
result = 0;
if (back)
{
- src = (const GFC_COMPLEX_8 * restrict) (((char*) base) + (len - 1) * delta);
+ src = PTR_ADD_OFFSET (base, (len - 1) * delta);
for (n = len; n > 0; n--)
{
if (*src == value)
@@ -325,7 +325,7 @@ mfindloc1_c8 (gfc_array_index_type * const restrict retarray,
result = 0;
if (back)
{
- src = (const GFC_COMPLEX_8 * restrict) (((char*)base) + (len - 1) * delta);
+ src = PTR_ADD_OFFSET (base, (len - 1) * delta);
msrc = mbase + (len - 1) * mdelta;
for (n = len; n > 0; n--)
{
diff --git a/libgfortran/generated/findloc1_i1.c b/libgfortran/generated/findloc1_i1.c
index 56de7e205772..7be40974e8a5 100644
--- a/libgfortran/generated/findloc1_i1.c
+++ b/libgfortran/generated/findloc1_i1.c
@@ -139,7 +139,7 @@ findloc1_i1 (gfc_array_index_type * const restrict retarray,
result = 0;
if (back)
{
- src = (const GFC_INTEGER_1 * restrict) (((char*) base) + (len - 1) * delta);
+ src = PTR_ADD_OFFSET (base, (len - 1) * delta);
for (n = len; n > 0; n--)
{
if (*src == value)
@@ -325,7 +325,7 @@ mfindloc1_i1 (gfc_array_index_type * const restrict retarray,
result = 0;
if (back)
{
- src = (const GFC_INTEGER_1 * restrict) (((char*)base) + (len - 1) * delta);
+ src = PTR_ADD_OFFSET (base, (len - 1) * delta);
msrc = mbase + (len - 1) * mdelta;
for (n = len; n > 0; n--)
{
diff --git a/libgfortran/generated/findloc1_i16.c b/libgfortran/generated/findloc1_i16.c
index 5917ebfc7910..909a618201fa 100644
--- a/libgfortran/generated/findloc1_i16.c
+++ b/libgfortran/generated/findloc1_i16.c
@@ -139,7 +139,7 @@ findloc1_i16 (gfc_array_index_type * const restrict retarray,
result = 0;
if (back)
{
- src = (const GFC_INTEGER_16 * restrict) (((char*) base) + (len - 1) * delta);
+ src = PTR_ADD_OFFSET (base, (len - 1) * delta);
for (n = len; n > 0; n--)
{
if (*src == value)
@@ -325,7 +325,7 @@ mfindloc1_i16 (gfc_array_index_type * const restrict retarray,
result = 0;
if (back)
{
- src = (const GFC_INTEGER_16 * restrict) (((char*)base) + (len - 1) * delta);
+ src = PTR_ADD_OFFSET (base, (len - 1) * delta);
msrc = mbase + (len - 1) * mdelta;
for (n = len; n > 0; n--)
{
diff --git a/libgfortran/generated/findloc1_i2.c b/libgfortran/generated/findloc1_i2.c
index bec080adf5a8..05f7604bf5c8 100644
--- a/libgfortran/generated/findloc1_i2.c
+++ b/libgfortran/generated/findloc1_i2.c
@@ -139,7 +139,7 @@ findloc1_i2 (gfc_array_index_type * const restrict retarray,
result = 0;
if (back)
{
- src = (const GFC_INTEGER_2 * restrict) (((char*) base) + (len - 1) * delta);
+ src = PTR_ADD_OFFSET (base, (len - 1) * delta);
for (n = len; n > 0; n--)
{
if (*src == value)
@@ -325,7 +325,7 @@ mfindloc1_i2 (gfc_array_index_type * const restrict retarray,
result = 0;
if (back)
{
- src = (const GFC_INTEGER_2 * restrict) (((char*)base) + (len - 1) * delta);
+ src = PTR_ADD_OFFSET (base, (len - 1) * delta);
msrc = mbase + (len - 1) * mdelta;
for (n = len; n > 0; n--)
{
diff --git a/libgfortran/generated/findloc1_i4.c b/libgfortran/generated/findloc1_i4.c
index c82c2c86b5b2..07954ffc581d 100644
--- a/libgfortran/generated/findloc1_i4.c
+++ b/libgfortran/generated/findloc1_i4.c
@@ -139,7 +139,7 @@ findloc1_i4 (gfc_array_index_type * const restrict retarray,
result = 0;
if (back)
{
- src = (const GFC_INTEGER_4 * restrict) (((char*) base) + (len - 1) * delta);
+ src = PTR_ADD_OFFSET (base, (len - 1) * delta);
for (n = len; n > 0; n--)
{
if (*src == value)
@@ -325,7 +325,7 @@ mfindloc1_i4 (gfc_array_index_type * const restrict retarray,
result = 0;
if (back)
{
- src = (const GFC_INTEGER_4 * restrict) (((char*)base) + (len - 1) * delta);
+ src = PTR_ADD_OFFSET (base, (len - 1) * delta);
msrc = mbase + (len - 1) * mdelta;
for (n = len; n > 0; n--)
{
diff --git a/libgfortran/generated/findloc1_i8.c b/libgfortran/generated/findloc1_i8.c
index 3c69005935d0..ed3f62b99f75 100644
--- a/libgfortran/generated/findloc1_i8.c
+++ b/libgfortran/generated/findloc1_i8.c
@@ -139,7 +139,7 @@ findloc1_i8 (gfc_array_index_type * const restrict retarray,
result = 0;
if (back)
{
- src = (const GFC_INTEGER_8 * restrict) (((char*) base) + (len - 1) * delta);
+ src = PTR_ADD_OFFSET (base, (len - 1) * delta);
for (n = len; n > 0; n--)
{
if (*src == value)
@@ -325,7 +325,7 @@ mfindloc1_i8 (gfc_array_index_type * const restrict retarray,
result = 0;
if (back)
{
- src = (const GFC_INTEGER_8 * restrict) (((char*)base) + (len - 1) * delta);
+ src = PTR_ADD_OFFSET (base, (len - 1) * delta);
msrc = mbase + (len - 1) * mdelta;
for (n = len; n > 0; n--)
{
diff --git a/libgfortran/generated/findloc1_r10.c b/libgfortran/generated/findloc1_r10.c
index a3db95012cb0..738268678301 100644
--- a/libgfortran/generated/findloc1_r10.c
+++ b/libgfortran/generated/findloc1_r10.c
@@ -139,7 +139,7 @@ findloc1_r10 (gfc_array_index_type * const restrict retarray,
result = 0;
if (back)
{
- src = (const GFC_REAL_10 * restrict) (((char*) base) + (len - 1) * delta);
+ src = PTR_ADD_OFFSET (base, (len - 1) * delta);
for (n = len; n > 0; n--)
{
if (*src == value)
@@ -325,7 +325,7 @@ mfindloc1_r10 (gfc_array_index_type * const restrict retarray,
result = 0;
if (back)
{
- src = (const GFC_REAL_10 * restrict) (((char*)base) + (len - 1) * delta);
+ src = PTR_ADD_OFFSET (base, (len - 1) * delta);
msrc = mbase + (len - 1) * mdelta;
for (n = len; n > 0; n--)
{
diff --git a/libgfortran/generated/findloc1_r16.c b/libgfortran/generated/findloc1_r16.c
index 0c29d3a9582c..7f9fa5ecfc3f 100644
--- a/libgfortran/generated/findloc1_r16.c
+++ b/libgfortran/generated/findloc1_r16.c
@@ -139,7 +139,7 @@ findloc1_r16 (gfc_array_index_type * const restrict retarray,
result = 0;
if (back)
{
- src = (const GFC_REAL_16 * restrict) (((char*) base) + (len - 1) * delta);
+ src = PTR_ADD_OFFSET (base, (len - 1) * delta);
for (n = len; n > 0; n--)
{
if (*src == value)
@@ -325,7 +325,7 @@ mfindloc1_r16 (gfc_array_index_type * const restrict retarray,
result = 0;
if (back)
{
- src = (const GFC_REAL_16 * restrict) (((char*)base) + (len - 1) * delta);
+ src = PTR_ADD_OFFSET (base, (len - 1) * delta);
msrc = mbase + (len - 1) * mdelta;
for (n = len; n > 0; n--)
{
diff --git a/libgfortran/generated/findloc1_r17.c b/libgfortran/generated/findloc1_r17.c
index 2a5711a0fbbd..95a79ac93aff 100644
--- a/libgfortran/generated/findloc1_r17.c
+++ b/libgfortran/generated/findloc1_r17.c
@@ -139,7 +139,7 @@ findloc1_r17 (gfc_array_index_type * const restrict retarray,
result = 0;
if (back)
{
- src = (const GFC_REAL_17 * restrict) (((char*) base) + (len - 1) * delta);
+ src = PTR_ADD_OFFSET (base, (len - 1) * delta);
for (n = len; n > 0; n--)
{
if (*src == value)
@@ -325,7 +325,7 @@ mfindloc1_r17 (gfc_array_index_type * const restrict retarray,
result = 0;
if (back)
{
- src = (const GFC_REAL_17 * restrict) (((char*)base) + (len - 1) * delta);
+ src = PTR_ADD_OFFSET (base, (len - 1) * delta);
msrc = mbase + (len - 1) * mdelta;
for (n = len; n > 0; n--)
{
diff --git a/libgfortran/generated/findloc1_r4.c b/libgfortran/generated/findloc1_r4.c
index 0bb9504951fc..f9d3f59341c9 100644
--- a/libgfortran/generated/findloc1_r4.c
+++ b/libgfortran/generated/findloc1_r4.c
@@ -139,7 +139,7 @@ findloc1_r4 (gfc_array_index_type * const restrict retarray,
result = 0;
if (back)
{
- src = (const GFC_REAL_4 * restrict) (((char*) base) + (len - 1) * delta);
+ src = PTR_ADD_OFFSET (base, (len - 1) * delta);
for (n = len; n > 0; n--)
{
if (*src == value)
@@ -325,7 +325,7 @@ mfindloc1_r4 (gfc_array_index_type * const restrict retarray,
result = 0;
if (back)
{
- src = (const GFC_REAL_4 * restrict) (((char*)base) + (len - 1) * delta);
+ src = PTR_ADD_OFFSET (base, (len - 1) * delta);
msrc = mbase + (len - 1) * mdelta;
for (n = len; n > 0; n--)
{
diff --git a/libgfortran/generated/findloc1_r8.c b/libgfortran/generated/findloc1_r8.c
index 3a24334f66b8..83ad7544d5f0 100644
--- a/libgfortran/generated/findloc1_r8.c
+++ b/libgfortran/generated/findloc1_r8.c
@@ -139,7 +139,7 @@ findloc1_r8 (gfc_array_index_type * const restrict retarray,
result = 0;
if (back)
{
- src = (const GFC_REAL_8 * restrict) (((char*) base) + (len - 1) * delta);
+ src = PTR_ADD_OFFSET (base, (len - 1) * delta);
for (n = len; n > 0; n--)
{
if (*src == value)
@@ -325,7 +325,7 @@ mfindloc1_r8 (gfc_array_index_type * const restrict retarray,
result = 0;
if (back)
{
- src = (const GFC_REAL_8 * restrict) (((char*)base) + (len - 1) * delta);
+ src = PTR_ADD_OFFSET (base, (len - 1) * delta);
msrc = mbase + (len - 1) * mdelta;
for (n = len; n > 0; n--)
{
diff --git a/libgfortran/generated/findloc1_s1.c b/libgfortran/generated/findloc1_s1.c
index f6979b9c73dd..a243e4606ac6 100644
--- a/libgfortran/generated/findloc1_s1.c
+++ b/libgfortran/generated/findloc1_s1.c
@@ -141,7 +141,7 @@ findloc1_s1 (gfc_array_index_type * const restrict retarray,
result = 0;
if (back)
{
- src = (const GFC_UINTEGER_1 * restrict) (((char*) base) + (len - 1) * delta);
+ src = PTR_ADD_OFFSET (base, (len - 1) * delta);
for (n = len; n > 0; n--)
{
if (compare_string (len_array, (char *) src, len_value, (char *) value) == 0)
@@ -327,7 +327,7 @@ mfindloc1_s1 (gfc_array_index_type * const restrict retarray,
result = 0;
if (back)
{
- src = (const GFC_UINTEGER_1 * restrict) (((char*)base) + (len - 1) * delta);
+ src = PTR_ADD_OFFSET (base, (len - 1) * delta);
msrc = mbase + (len - 1) * mdelta;
for (n = len; n > 0; n--)
{
diff --git a/libgfortran/generated/findloc1_s4.c b/libgfortran/generated/findloc1_s4.c
index 652aa79abbfd..4214658bd007 100644
--- a/libgfortran/generated/findloc1_s4.c
+++ b/libgfortran/generated/findloc1_s4.c
@@ -141,7 +141,7 @@ findloc1_s4 (gfc_array_index_type * const restrict retarray,
result = 0;
if (back)
{
- src = (const GFC_UINTEGER_4 * restrict) (((char*) base) + (len - 1) * delta);
+ src = PTR_ADD_OFFSET (base, (len - 1) * delta);
for (n = len; n > 0; n--)
{
if (compare_string_char4 (len_array, src, len_value, value) == 0)
@@ -327,7 +327,7 @@ mfindloc1_s4 (gfc_array_index_type * const restrict retarray,
result = 0;
if (back)
{
- src = (const GFC_UINTEGER_4 * restrict) (((char*)base) + (len - 1) * delta);
+ src = PTR_ADD_OFFSET (base, (len - 1) * delta);
msrc = mbase + (len - 1) * mdelta;
for (n = len; n > 0; n--)
{
diff --git a/libgfortran/generated/findloc2_s1.c b/libgfortran/generated/findloc2_s1.c
index 70c49d33295d..5f4655b1d794 100644
--- a/libgfortran/generated/findloc2_s1.c
+++ b/libgfortran/generated/findloc2_s1.c
@@ -48,7 +48,7 @@ findloc2_s1 (gfc_array_s1 * const restrict array, const GFC_UINTEGER_1 * restric
sstride = GFC_DESCRIPTOR_STRIDE_BYTES(array,0);
if (back)
{
- src = (const GFC_UINTEGER_1 * restrict) (((char*)array->base_addr) + (extent - 1) * sstride);
+ src = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, extent - 1);
for (i = extent; i > 0; i--)
{
if (compare_string (len_array, (char *) src, len_value, (char *) value) == 0)
@@ -110,7 +110,7 @@ mfindloc2_s1 (gfc_array_s1 * const restrict array,
if (back)
{
- src = (const GFC_UINTEGER_1 * restrict) (((char*)array->base_addr) + (extent - 1) * sstride);
+ src = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, extent - 1);
mbase += (extent - 1) * mstride;
for (i = extent; i > 0; i--)
{
diff --git a/libgfortran/generated/findloc2_s4.c b/libgfortran/generated/findloc2_s4.c
index 8f8f09660364..d9f96b92ccbc 100644
--- a/libgfortran/generated/findloc2_s4.c
+++ b/libgfortran/generated/findloc2_s4.c
@@ -48,7 +48,7 @@ findloc2_s4 (gfc_array_s4 * const restrict array, const GFC_UINTEGER_4 * restric
sstride = GFC_DESCRIPTOR_STRIDE_BYTES(array,0);
if (back)
{
- src = (const GFC_UINTEGER_4 * restrict) (((char*)array->base_addr) + (extent - 1) * sstride);
+ src = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, extent - 1);
for (i = extent; i > 0; i--)
{
if (compare_string_char4 (len_array, src, len_value, value) == 0)
@@ -110,7 +110,7 @@ mfindloc2_s4 (gfc_array_s4 * const restrict array,
if (back)
{
- src = (const GFC_UINTEGER_4 * restrict) (((char*)array->base_addr) + (extent - 1) * sstride);
+ src = GFC_DESCRIPTOR1_ELEM_ADDRESS (array, extent - 1);
mbase += (extent - 1) * mstride;
for (i = extent; i > 0; i--)
{
diff --git a/libgfortran/generated/matmul_c10.c b/libgfortran/generated/matmul_c10.c
index 3d0780fa880c..f5c298aa0c4f 100644
--- a/libgfortran/generated/matmul_c10.c
+++ b/libgfortran/generated/matmul_c10.c
@@ -94,7 +94,8 @@ matmul_c10_avx (gfc_array_c10 * const restrict retarray,
index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
index_type x, y, n, count, xcount, ycount;
- index_type aystride_bytes, bystride_bytes, rystride_bytes;
+ index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
+ rxstride_bytes, rystride_bytes;
assert (GFC_DESCRIPTOR_RANK (a) == 2
|| GFC_DESCRIPTOR_RANK (b) == 2);
@@ -183,12 +184,13 @@ matmul_c10_avx (gfc_array_c10 * const restrict retarray,
either as a row or a column matrix. We want both cases to
work. */
rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
+ rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
}
else
{
rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
+ rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
}
@@ -197,6 +199,7 @@ matmul_c10_avx (gfc_array_c10 * const restrict retarray,
/* Treat it as a a row matrix A[1,count]. */
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = 1;
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = sizeof (GFC_COMPLEX_10);
xcount = 1;
@@ -206,6 +209,7 @@ matmul_c10_avx (gfc_array_c10 * const restrict retarray,
{
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = GFC_DESCRIPTOR_STRIDE(a,1);
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,1);
count = GFC_DESCRIPTOR_EXTENT(a,1);
@@ -224,17 +228,20 @@ matmul_c10_avx (gfc_array_c10 * const restrict retarray,
{
/* Treat it as a column matrix B[count,1] */
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
/* bystride should never be used for 1-dimensional b.
The value is only used for calculation of the
memory by the buffer. */
bystride = 256;
+ bystride_bytes = 99999999;
ycount = 1;
}
else
{
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
bystride = GFC_DESCRIPTOR_STRIDE(b,1);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
bystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,1);
ycount = GFC_DESCRIPTOR_EXTENT(b,1);
}
@@ -294,12 +301,11 @@ matmul_c10_avx (gfc_array_c10 * const restrict retarray,
from netlib.org, translated to C, and modified for matmul.m4. */
- const GFC_COMPLEX_10 *a, *b;
GFC_COMPLEX_10 *c;
const index_type m = xcount, n = ycount, k = count;
/* System generated locals */
- index_type a_dim1, b_dim1, c_dim1,
+ index_type a_dim1, b_dim1,
i1, i2, i3, i4, i5, i6;
/* Local variables */
@@ -309,19 +315,25 @@ matmul_c10_avx (gfc_array_c10 * const restrict retarray,
index_type isec, jsec, lsec, uisec, ujsec, ulsec;
GFC_COMPLEX_10 *t1;
- a = abase;
- b = bbase;
c = retarray->base_addr;
/* Parameter adjustments */
- c_dim1 = rystride;
a_dim1 = aystride;
b_dim1 = bystride;
- /* Empty c first. */
+#define A_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (abase, (i) * sizeof (GFC_COMPLEX_10) + (j) * aystride_bytes))
+
+#define B_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (bbase, (i) * sizeof (GFC_COMPLEX_10) + (j) * bystride_bytes))
+
+#define C_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (c, (i) * sizeof (GFC_COMPLEX_10) + (j) * rystride_bytes))
+
+ /* Empty result first. */
for (j=0; j<n; j++)
for (i=0; i<m; i++)
- c[i + j * c_dim1] = (GFC_COMPLEX_10)0;
+ C_ARRAY_ELEM (i, j) = (GFC_COMPLEX_10)0;
/* Early exit if possible */
if (m == 0 || n == 0 || k == 0)
@@ -373,20 +385,20 @@ matmul_c10_avx (gfc_array_c10 * const restrict retarray,
for (i = ii; i < i5; i += 2)
{
t1[l - ll + 1 + ((i - ii + 1) << 8) - 257] =
- a[i + l * a_dim1];
+ A_ARRAY_ELEM (i, l);
t1[l - ll + 2 + ((i - ii + 1) << 8) - 257] =
- a[i + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i, l + 1);
t1[l - ll + 1 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + l * a_dim1];
+ A_ARRAY_ELEM (i + 1, l);
t1[l - ll + 2 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i + 1, l + 1);
}
if (uisec < isec)
{
t1[l - ll + 1 + (isec << 8) - 257] =
- a[ii + isec - 1 + l * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l);
t1[l - ll + 2 + (isec << 8) - 257] =
- a[ii + isec - 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l + 1);
}
}
if (ulsec < lsec)
@@ -395,7 +407,7 @@ matmul_c10_avx (gfc_array_c10 * const restrict retarray,
for (i = ii; i< i4; ++i)
{
t1[lsec + ((i - ii + 1) << 8) - 257] =
- a[i + (ll + lsec - 1) * a_dim1];
+ A_ARRAY_ELEM (i, ll + lsec - 1);
}
}
@@ -406,100 +418,100 @@ matmul_c10_avx (gfc_array_c10 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f22 = c[i + 1 + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f23 = c[i + 1 + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
- f24 = c[i + 1 + (j + 3) * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
- f32 = c[i + 2 + (j + 1) * c_dim1];
- f42 = c[i + 3 + (j + 1) * c_dim1];
- f33 = c[i + 2 + (j + 2) * c_dim1];
- f43 = c[i + 3 + (j + 2) * c_dim1];
- f34 = c[i + 2 + (j + 3) * c_dim1];
- f44 = c[i + 3 + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f22 = C_ARRAY_ELEM (i + 1, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f23 = C_ARRAY_ELEM (i + 1, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
+ f24 = C_ARRAY_ELEM (i + 1, j + 3);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
+ f32 = C_ARRAY_ELEM (i + 2, j + 1);
+ f42 = C_ARRAY_ELEM (i + 3, j + 1);
+ f33 = C_ARRAY_ELEM (i + 2, j + 2);
+ f43 = C_ARRAY_ELEM (i + 3, j + 2);
+ f34 = C_ARRAY_ELEM (i + 2, j + 3);
+ f44 = C_ARRAY_ELEM (i + 3, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f22 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f23 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f24 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f32 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f42 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f33 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f43 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f34 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f44 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + 1 + (j + 1) * c_dim1] = f22;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + 1 + (j + 2) * c_dim1] = f23;
- c[i + (j + 3) * c_dim1] = f14;
- c[i + 1 + (j + 3) * c_dim1] = f24;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
- c[i + 2 + (j + 1) * c_dim1] = f32;
- c[i + 3 + (j + 1) * c_dim1] = f42;
- c[i + 2 + (j + 2) * c_dim1] = f33;
- c[i + 3 + (j + 2) * c_dim1] = f43;
- c[i + 2 + (j + 3) * c_dim1] = f34;
- c[i + 3 + (j + 3) * c_dim1] = f44;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i + 1, j + 1) = f22;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i + 1, j + 2) = f23;
+ C_ARRAY_ELEM (i, j + 3) = f14;
+ C_ARRAY_ELEM (i + 1, j + 3) = f24;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
+ C_ARRAY_ELEM (i + 2, j + 1) = f32;
+ C_ARRAY_ELEM (i + 3, j + 1) = f42;
+ C_ARRAY_ELEM (i + 2, j + 2) = f33;
+ C_ARRAY_ELEM (i + 3, j + 2) = f43;
+ C_ARRAY_ELEM (i + 2, j + 3) = f34;
+ C_ARRAY_ELEM (i + 3, j + 3) = f44;
}
if (uisec < isec)
{
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 1) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 2) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 3) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + (j + 3) * c_dim1] = f14;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i, j + 3) = f14;
}
}
}
@@ -511,38 +523,38 @@ matmul_c10_avx (gfc_array_c10 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
}
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
+ C_ARRAY_ELEM (i, j) = f11;
}
}
}
@@ -551,6 +563,9 @@ matmul_c10_avx (gfc_array_c10 * const restrict retarray,
}
free(t1);
return;
+#undef A_ARRAY_ELEM
+#undef B_ARRAY_ELEM
+#undef C_ARRAY_ELEM
}
else if (rxstride == 1 && aystride == 1 && bxstride == 1)
{
@@ -563,11 +578,11 @@ matmul_c10_avx (gfc_array_c10 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_10) 0;
for (n = 0; n < count; n++)
s += abase_x[n] * bbase_y[n];
@@ -582,11 +597,11 @@ matmul_c10_avx (gfc_array_c10 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_10) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n];
- dest[y*rystride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n) * bbase_y[n];
+ ARRAY_ELEM_AT_OFFSET (dest, y * rystride_bytes) = s;
}
}
}
@@ -597,26 +612,27 @@ matmul_c10_avx (gfc_array_c10 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_10) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n*bxstride];
- dest[y*rxstride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
}
}
else if (axstride < aystride)
{
for (y = 0; y < ycount; y++)
for (x = 0; x < xcount; x++)
- dest[x*rxstride + y*rystride] = (GFC_COMPLEX_10)0;
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y) = (GFC_COMPLEX_10)0;
for (y = 0; y < ycount; y++)
for (n = 0; n < count; n++)
for (x = 0; x < xcount; x++)
/* dest[x,y] += a[x,n] * b[n,y] */
- dest[x*rxstride + y*rystride] +=
- abase[x*axstride + n*aystride] *
- bbase[n*bxstride + y*bystride];
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y)
+ += GFC_DESCRIPTOR2_ELEM (a, x, n)
+ * GFC_DESCRIPTOR2_ELEM (b, n, y);
}
else
{
@@ -627,15 +643,16 @@ matmul_c10_avx (gfc_array_c10 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_10) 0;
for (n = 0; n < count; n++)
- s += abase_x[n*aystride] * bbase_y[n*bxstride];
- dest_y[x*rxstride] = s;
+ s += ARRAY_ELEM_AT_OFFSET (abase_x, n * aystride_bytes)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ ARRAY_ELEM_AT_OFFSET (dest_y, x * rxstride_bytes) = s;
}
}
}
@@ -662,7 +679,8 @@ matmul_c10_avx2 (gfc_array_c10 * const restrict retarray,
index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
index_type x, y, n, count, xcount, ycount;
- index_type aystride_bytes, bystride_bytes, rystride_bytes;
+ index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
+ rxstride_bytes, rystride_bytes;
assert (GFC_DESCRIPTOR_RANK (a) == 2
|| GFC_DESCRIPTOR_RANK (b) == 2);
@@ -751,12 +769,13 @@ matmul_c10_avx2 (gfc_array_c10 * const restrict retarray,
either as a row or a column matrix. We want both cases to
work. */
rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
+ rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
}
else
{
rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
+ rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
}
@@ -765,6 +784,7 @@ matmul_c10_avx2 (gfc_array_c10 * const restrict retarray,
/* Treat it as a a row matrix A[1,count]. */
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = 1;
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = sizeof (GFC_COMPLEX_10);
xcount = 1;
@@ -774,6 +794,7 @@ matmul_c10_avx2 (gfc_array_c10 * const restrict retarray,
{
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = GFC_DESCRIPTOR_STRIDE(a,1);
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,1);
count = GFC_DESCRIPTOR_EXTENT(a,1);
@@ -792,17 +813,20 @@ matmul_c10_avx2 (gfc_array_c10 * const restrict retarray,
{
/* Treat it as a column matrix B[count,1] */
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
/* bystride should never be used for 1-dimensional b.
The value is only used for calculation of the
memory by the buffer. */
bystride = 256;
+ bystride_bytes = 99999999;
ycount = 1;
}
else
{
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
bystride = GFC_DESCRIPTOR_STRIDE(b,1);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
bystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,1);
ycount = GFC_DESCRIPTOR_EXTENT(b,1);
}
@@ -862,12 +886,11 @@ matmul_c10_avx2 (gfc_array_c10 * const restrict retarray,
from netlib.org, translated to C, and modified for matmul.m4. */
- const GFC_COMPLEX_10 *a, *b;
GFC_COMPLEX_10 *c;
const index_type m = xcount, n = ycount, k = count;
/* System generated locals */
- index_type a_dim1, b_dim1, c_dim1,
+ index_type a_dim1, b_dim1,
i1, i2, i3, i4, i5, i6;
/* Local variables */
@@ -877,19 +900,25 @@ matmul_c10_avx2 (gfc_array_c10 * const restrict retarray,
index_type isec, jsec, lsec, uisec, ujsec, ulsec;
GFC_COMPLEX_10 *t1;
- a = abase;
- b = bbase;
c = retarray->base_addr;
/* Parameter adjustments */
- c_dim1 = rystride;
a_dim1 = aystride;
b_dim1 = bystride;
- /* Empty c first. */
+#define A_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (abase, (i) * sizeof (GFC_COMPLEX_10) + (j) * aystride_bytes))
+
+#define B_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (bbase, (i) * sizeof (GFC_COMPLEX_10) + (j) * bystride_bytes))
+
+#define C_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (c, (i) * sizeof (GFC_COMPLEX_10) + (j) * rystride_bytes))
+
+ /* Empty result first. */
for (j=0; j<n; j++)
for (i=0; i<m; i++)
- c[i + j * c_dim1] = (GFC_COMPLEX_10)0;
+ C_ARRAY_ELEM (i, j) = (GFC_COMPLEX_10)0;
/* Early exit if possible */
if (m == 0 || n == 0 || k == 0)
@@ -941,20 +970,20 @@ matmul_c10_avx2 (gfc_array_c10 * const restrict retarray,
for (i = ii; i < i5; i += 2)
{
t1[l - ll + 1 + ((i - ii + 1) << 8) - 257] =
- a[i + l * a_dim1];
+ A_ARRAY_ELEM (i, l);
t1[l - ll + 2 + ((i - ii + 1) << 8) - 257] =
- a[i + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i, l + 1);
t1[l - ll + 1 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + l * a_dim1];
+ A_ARRAY_ELEM (i + 1, l);
t1[l - ll + 2 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i + 1, l + 1);
}
if (uisec < isec)
{
t1[l - ll + 1 + (isec << 8) - 257] =
- a[ii + isec - 1 + l * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l);
t1[l - ll + 2 + (isec << 8) - 257] =
- a[ii + isec - 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l + 1);
}
}
if (ulsec < lsec)
@@ -963,7 +992,7 @@ matmul_c10_avx2 (gfc_array_c10 * const restrict retarray,
for (i = ii; i< i4; ++i)
{
t1[lsec + ((i - ii + 1) << 8) - 257] =
- a[i + (ll + lsec - 1) * a_dim1];
+ A_ARRAY_ELEM (i, ll + lsec - 1);
}
}
@@ -974,100 +1003,100 @@ matmul_c10_avx2 (gfc_array_c10 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f22 = c[i + 1 + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f23 = c[i + 1 + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
- f24 = c[i + 1 + (j + 3) * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
- f32 = c[i + 2 + (j + 1) * c_dim1];
- f42 = c[i + 3 + (j + 1) * c_dim1];
- f33 = c[i + 2 + (j + 2) * c_dim1];
- f43 = c[i + 3 + (j + 2) * c_dim1];
- f34 = c[i + 2 + (j + 3) * c_dim1];
- f44 = c[i + 3 + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f22 = C_ARRAY_ELEM (i + 1, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f23 = C_ARRAY_ELEM (i + 1, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
+ f24 = C_ARRAY_ELEM (i + 1, j + 3);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
+ f32 = C_ARRAY_ELEM (i + 2, j + 1);
+ f42 = C_ARRAY_ELEM (i + 3, j + 1);
+ f33 = C_ARRAY_ELEM (i + 2, j + 2);
+ f43 = C_ARRAY_ELEM (i + 3, j + 2);
+ f34 = C_ARRAY_ELEM (i + 2, j + 3);
+ f44 = C_ARRAY_ELEM (i + 3, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f22 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f23 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f24 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f32 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f42 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f33 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f43 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f34 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f44 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + 1 + (j + 1) * c_dim1] = f22;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + 1 + (j + 2) * c_dim1] = f23;
- c[i + (j + 3) * c_dim1] = f14;
- c[i + 1 + (j + 3) * c_dim1] = f24;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
- c[i + 2 + (j + 1) * c_dim1] = f32;
- c[i + 3 + (j + 1) * c_dim1] = f42;
- c[i + 2 + (j + 2) * c_dim1] = f33;
- c[i + 3 + (j + 2) * c_dim1] = f43;
- c[i + 2 + (j + 3) * c_dim1] = f34;
- c[i + 3 + (j + 3) * c_dim1] = f44;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i + 1, j + 1) = f22;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i + 1, j + 2) = f23;
+ C_ARRAY_ELEM (i, j + 3) = f14;
+ C_ARRAY_ELEM (i + 1, j + 3) = f24;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
+ C_ARRAY_ELEM (i + 2, j + 1) = f32;
+ C_ARRAY_ELEM (i + 3, j + 1) = f42;
+ C_ARRAY_ELEM (i + 2, j + 2) = f33;
+ C_ARRAY_ELEM (i + 3, j + 2) = f43;
+ C_ARRAY_ELEM (i + 2, j + 3) = f34;
+ C_ARRAY_ELEM (i + 3, j + 3) = f44;
}
if (uisec < isec)
{
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 1) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 2) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 3) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + (j + 3) * c_dim1] = f14;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i, j + 3) = f14;
}
}
}
@@ -1079,38 +1108,38 @@ matmul_c10_avx2 (gfc_array_c10 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
}
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
+ C_ARRAY_ELEM (i, j) = f11;
}
}
}
@@ -1119,6 +1148,9 @@ matmul_c10_avx2 (gfc_array_c10 * const restrict retarray,
}
free(t1);
return;
+#undef A_ARRAY_ELEM
+#undef B_ARRAY_ELEM
+#undef C_ARRAY_ELEM
}
else if (rxstride == 1 && aystride == 1 && bxstride == 1)
{
@@ -1131,11 +1163,11 @@ matmul_c10_avx2 (gfc_array_c10 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_10) 0;
for (n = 0; n < count; n++)
s += abase_x[n] * bbase_y[n];
@@ -1150,11 +1182,11 @@ matmul_c10_avx2 (gfc_array_c10 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_10) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n];
- dest[y*rystride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n) * bbase_y[n];
+ ARRAY_ELEM_AT_OFFSET (dest, y * rystride_bytes) = s;
}
}
}
@@ -1165,26 +1197,27 @@ matmul_c10_avx2 (gfc_array_c10 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_10) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n*bxstride];
- dest[y*rxstride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
}
}
else if (axstride < aystride)
{
for (y = 0; y < ycount; y++)
for (x = 0; x < xcount; x++)
- dest[x*rxstride + y*rystride] = (GFC_COMPLEX_10)0;
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y) = (GFC_COMPLEX_10)0;
for (y = 0; y < ycount; y++)
for (n = 0; n < count; n++)
for (x = 0; x < xcount; x++)
/* dest[x,y] += a[x,n] * b[n,y] */
- dest[x*rxstride + y*rystride] +=
- abase[x*axstride + n*aystride] *
- bbase[n*bxstride + y*bystride];
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y)
+ += GFC_DESCRIPTOR2_ELEM (a, x, n)
+ * GFC_DESCRIPTOR2_ELEM (b, n, y);
}
else
{
@@ -1195,15 +1228,16 @@ matmul_c10_avx2 (gfc_array_c10 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_10) 0;
for (n = 0; n < count; n++)
- s += abase_x[n*aystride] * bbase_y[n*bxstride];
- dest_y[x*rxstride] = s;
+ s += ARRAY_ELEM_AT_OFFSET (abase_x, n * aystride_bytes)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ ARRAY_ELEM_AT_OFFSET (dest_y, x * rxstride_bytes) = s;
}
}
}
@@ -1230,7 +1264,8 @@ matmul_c10_avx512f (gfc_array_c10 * const restrict retarray,
index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
index_type x, y, n, count, xcount, ycount;
- index_type aystride_bytes, bystride_bytes, rystride_bytes;
+ index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
+ rxstride_bytes, rystride_bytes;
assert (GFC_DESCRIPTOR_RANK (a) == 2
|| GFC_DESCRIPTOR_RANK (b) == 2);
@@ -1319,12 +1354,13 @@ matmul_c10_avx512f (gfc_array_c10 * const restrict retarray,
either as a row or a column matrix. We want both cases to
work. */
rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
+ rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
}
else
{
rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
+ rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
}
@@ -1333,6 +1369,7 @@ matmul_c10_avx512f (gfc_array_c10 * const restrict retarray,
/* Treat it as a a row matrix A[1,count]. */
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = 1;
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = sizeof (GFC_COMPLEX_10);
xcount = 1;
@@ -1342,6 +1379,7 @@ matmul_c10_avx512f (gfc_array_c10 * const restrict retarray,
{
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = GFC_DESCRIPTOR_STRIDE(a,1);
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,1);
count = GFC_DESCRIPTOR_EXTENT(a,1);
@@ -1360,17 +1398,20 @@ matmul_c10_avx512f (gfc_array_c10 * const restrict retarray,
{
/* Treat it as a column matrix B[count,1] */
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
/* bystride should never be used for 1-dimensional b.
The value is only used for calculation of the
memory by the buffer. */
bystride = 256;
+ bystride_bytes = 99999999;
ycount = 1;
}
else
{
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
bystride = GFC_DESCRIPTOR_STRIDE(b,1);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
bystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,1);
ycount = GFC_DESCRIPTOR_EXTENT(b,1);
}
@@ -1430,12 +1471,11 @@ matmul_c10_avx512f (gfc_array_c10 * const restrict retarray,
from netlib.org, translated to C, and modified for matmul.m4. */
- const GFC_COMPLEX_10 *a, *b;
GFC_COMPLEX_10 *c;
const index_type m = xcount, n = ycount, k = count;
/* System generated locals */
- index_type a_dim1, b_dim1, c_dim1,
+ index_type a_dim1, b_dim1,
i1, i2, i3, i4, i5, i6;
/* Local variables */
@@ -1445,19 +1485,25 @@ matmul_c10_avx512f (gfc_array_c10 * const restrict retarray,
index_type isec, jsec, lsec, uisec, ujsec, ulsec;
GFC_COMPLEX_10 *t1;
- a = abase;
- b = bbase;
c = retarray->base_addr;
/* Parameter adjustments */
- c_dim1 = rystride;
a_dim1 = aystride;
b_dim1 = bystride;
- /* Empty c first. */
+#define A_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (abase, (i) * sizeof (GFC_COMPLEX_10) + (j) * aystride_bytes))
+
+#define B_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (bbase, (i) * sizeof (GFC_COMPLEX_10) + (j) * bystride_bytes))
+
+#define C_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (c, (i) * sizeof (GFC_COMPLEX_10) + (j) * rystride_bytes))
+
+ /* Empty result first. */
for (j=0; j<n; j++)
for (i=0; i<m; i++)
- c[i + j * c_dim1] = (GFC_COMPLEX_10)0;
+ C_ARRAY_ELEM (i, j) = (GFC_COMPLEX_10)0;
/* Early exit if possible */
if (m == 0 || n == 0 || k == 0)
@@ -1509,20 +1555,20 @@ matmul_c10_avx512f (gfc_array_c10 * const restrict retarray,
for (i = ii; i < i5; i += 2)
{
t1[l - ll + 1 + ((i - ii + 1) << 8) - 257] =
- a[i + l * a_dim1];
+ A_ARRAY_ELEM (i, l);
t1[l - ll + 2 + ((i - ii + 1) << 8) - 257] =
- a[i + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i, l + 1);
t1[l - ll + 1 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + l * a_dim1];
+ A_ARRAY_ELEM (i + 1, l);
t1[l - ll + 2 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i + 1, l + 1);
}
if (uisec < isec)
{
t1[l - ll + 1 + (isec << 8) - 257] =
- a[ii + isec - 1 + l * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l);
t1[l - ll + 2 + (isec << 8) - 257] =
- a[ii + isec - 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l + 1);
}
}
if (ulsec < lsec)
@@ -1531,7 +1577,7 @@ matmul_c10_avx512f (gfc_array_c10 * const restrict retarray,
for (i = ii; i< i4; ++i)
{
t1[lsec + ((i - ii + 1) << 8) - 257] =
- a[i + (ll + lsec - 1) * a_dim1];
+ A_ARRAY_ELEM (i, ll + lsec - 1);
}
}
@@ -1542,100 +1588,100 @@ matmul_c10_avx512f (gfc_array_c10 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f22 = c[i + 1 + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f23 = c[i + 1 + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
- f24 = c[i + 1 + (j + 3) * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
- f32 = c[i + 2 + (j + 1) * c_dim1];
- f42 = c[i + 3 + (j + 1) * c_dim1];
- f33 = c[i + 2 + (j + 2) * c_dim1];
- f43 = c[i + 3 + (j + 2) * c_dim1];
- f34 = c[i + 2 + (j + 3) * c_dim1];
- f44 = c[i + 3 + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f22 = C_ARRAY_ELEM (i + 1, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f23 = C_ARRAY_ELEM (i + 1, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
+ f24 = C_ARRAY_ELEM (i + 1, j + 3);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
+ f32 = C_ARRAY_ELEM (i + 2, j + 1);
+ f42 = C_ARRAY_ELEM (i + 3, j + 1);
+ f33 = C_ARRAY_ELEM (i + 2, j + 2);
+ f43 = C_ARRAY_ELEM (i + 3, j + 2);
+ f34 = C_ARRAY_ELEM (i + 2, j + 3);
+ f44 = C_ARRAY_ELEM (i + 3, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f22 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f23 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f24 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f32 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f42 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f33 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f43 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f34 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f44 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + 1 + (j + 1) * c_dim1] = f22;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + 1 + (j + 2) * c_dim1] = f23;
- c[i + (j + 3) * c_dim1] = f14;
- c[i + 1 + (j + 3) * c_dim1] = f24;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
- c[i + 2 + (j + 1) * c_dim1] = f32;
- c[i + 3 + (j + 1) * c_dim1] = f42;
- c[i + 2 + (j + 2) * c_dim1] = f33;
- c[i + 3 + (j + 2) * c_dim1] = f43;
- c[i + 2 + (j + 3) * c_dim1] = f34;
- c[i + 3 + (j + 3) * c_dim1] = f44;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i + 1, j + 1) = f22;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i + 1, j + 2) = f23;
+ C_ARRAY_ELEM (i, j + 3) = f14;
+ C_ARRAY_ELEM (i + 1, j + 3) = f24;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
+ C_ARRAY_ELEM (i + 2, j + 1) = f32;
+ C_ARRAY_ELEM (i + 3, j + 1) = f42;
+ C_ARRAY_ELEM (i + 2, j + 2) = f33;
+ C_ARRAY_ELEM (i + 3, j + 2) = f43;
+ C_ARRAY_ELEM (i + 2, j + 3) = f34;
+ C_ARRAY_ELEM (i + 3, j + 3) = f44;
}
if (uisec < isec)
{
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 1) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 2) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 3) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + (j + 3) * c_dim1] = f14;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i, j + 3) = f14;
}
}
}
@@ -1647,38 +1693,38 @@ matmul_c10_avx512f (gfc_array_c10 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
}
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
+ C_ARRAY_ELEM (i, j) = f11;
}
}
}
@@ -1687,6 +1733,9 @@ matmul_c10_avx512f (gfc_array_c10 * const restrict retarray,
}
free(t1);
return;
+#undef A_ARRAY_ELEM
+#undef B_ARRAY_ELEM
+#undef C_ARRAY_ELEM
}
else if (rxstride == 1 && aystride == 1 && bxstride == 1)
{
@@ -1699,11 +1748,11 @@ matmul_c10_avx512f (gfc_array_c10 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_10) 0;
for (n = 0; n < count; n++)
s += abase_x[n] * bbase_y[n];
@@ -1718,11 +1767,11 @@ matmul_c10_avx512f (gfc_array_c10 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_10) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n];
- dest[y*rystride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n) * bbase_y[n];
+ ARRAY_ELEM_AT_OFFSET (dest, y * rystride_bytes) = s;
}
}
}
@@ -1733,26 +1782,27 @@ matmul_c10_avx512f (gfc_array_c10 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_10) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n*bxstride];
- dest[y*rxstride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
}
}
else if (axstride < aystride)
{
for (y = 0; y < ycount; y++)
for (x = 0; x < xcount; x++)
- dest[x*rxstride + y*rystride] = (GFC_COMPLEX_10)0;
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y) = (GFC_COMPLEX_10)0;
for (y = 0; y < ycount; y++)
for (n = 0; n < count; n++)
for (x = 0; x < xcount; x++)
/* dest[x,y] += a[x,n] * b[n,y] */
- dest[x*rxstride + y*rystride] +=
- abase[x*axstride + n*aystride] *
- bbase[n*bxstride + y*bystride];
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y)
+ += GFC_DESCRIPTOR2_ELEM (a, x, n)
+ * GFC_DESCRIPTOR2_ELEM (b, n, y);
}
else
{
@@ -1763,15 +1813,16 @@ matmul_c10_avx512f (gfc_array_c10 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_10) 0;
for (n = 0; n < count; n++)
- s += abase_x[n*aystride] * bbase_y[n*bxstride];
- dest_y[x*rxstride] = s;
+ s += ARRAY_ELEM_AT_OFFSET (abase_x, n * aystride_bytes)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ ARRAY_ELEM_AT_OFFSET (dest_y, x * rxstride_bytes) = s;
}
}
}
@@ -1812,7 +1863,8 @@ matmul_c10_vanilla (gfc_array_c10 * const restrict retarray,
index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
index_type x, y, n, count, xcount, ycount;
- index_type aystride_bytes, bystride_bytes, rystride_bytes;
+ index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
+ rxstride_bytes, rystride_bytes;
assert (GFC_DESCRIPTOR_RANK (a) == 2
|| GFC_DESCRIPTOR_RANK (b) == 2);
@@ -1901,12 +1953,13 @@ matmul_c10_vanilla (gfc_array_c10 * const restrict retarray,
either as a row or a column matrix. We want both cases to
work. */
rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
+ rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
}
else
{
rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
+ rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
}
@@ -1915,6 +1968,7 @@ matmul_c10_vanilla (gfc_array_c10 * const restrict retarray,
/* Treat it as a a row matrix A[1,count]. */
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = 1;
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = sizeof (GFC_COMPLEX_10);
xcount = 1;
@@ -1924,6 +1978,7 @@ matmul_c10_vanilla (gfc_array_c10 * const restrict retarray,
{
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = GFC_DESCRIPTOR_STRIDE(a,1);
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,1);
count = GFC_DESCRIPTOR_EXTENT(a,1);
@@ -1942,17 +1997,20 @@ matmul_c10_vanilla (gfc_array_c10 * const restrict retarray,
{
/* Treat it as a column matrix B[count,1] */
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
/* bystride should never be used for 1-dimensional b.
The value is only used for calculation of the
memory by the buffer. */
bystride = 256;
+ bystride_bytes = 99999999;
ycount = 1;
}
else
{
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
bystride = GFC_DESCRIPTOR_STRIDE(b,1);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
bystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,1);
ycount = GFC_DESCRIPTOR_EXTENT(b,1);
}
@@ -2012,12 +2070,11 @@ matmul_c10_vanilla (gfc_array_c10 * const restrict retarray,
from netlib.org, translated to C, and modified for matmul.m4. */
- const GFC_COMPLEX_10 *a, *b;
GFC_COMPLEX_10 *c;
const index_type m = xcount, n = ycount, k = count;
/* System generated locals */
- index_type a_dim1, b_dim1, c_dim1,
+ index_type a_dim1, b_dim1,
i1, i2, i3, i4, i5, i6;
/* Local variables */
@@ -2027,19 +2084,25 @@ matmul_c10_vanilla (gfc_array_c10 * const restrict retarray,
index_type isec, jsec, lsec, uisec, ujsec, ulsec;
GFC_COMPLEX_10 *t1;
- a = abase;
- b = bbase;
c = retarray->base_addr;
/* Parameter adjustments */
- c_dim1 = rystride;
a_dim1 = aystride;
b_dim1 = bystride;
- /* Empty c first. */
+#define A_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (abase, (i) * sizeof (GFC_COMPLEX_10) + (j) * aystride_bytes))
+
+#define B_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (bbase, (i) * sizeof (GFC_COMPLEX_10) + (j) * bystride_bytes))
+
+#define C_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (c, (i) * sizeof (GFC_COMPLEX_10) + (j) * rystride_bytes))
+
+ /* Empty result first. */
for (j=0; j<n; j++)
for (i=0; i<m; i++)
- c[i + j * c_dim1] = (GFC_COMPLEX_10)0;
+ C_ARRAY_ELEM (i, j) = (GFC_COMPLEX_10)0;
/* Early exit if possible */
if (m == 0 || n == 0 || k == 0)
@@ -2091,20 +2154,20 @@ matmul_c10_vanilla (gfc_array_c10 * const restrict retarray,
for (i = ii; i < i5; i += 2)
{
t1[l - ll + 1 + ((i - ii + 1) << 8) - 257] =
- a[i + l * a_dim1];
+ A_ARRAY_ELEM (i, l);
t1[l - ll + 2 + ((i - ii + 1) << 8) - 257] =
- a[i + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i, l + 1);
t1[l - ll + 1 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + l * a_dim1];
+ A_ARRAY_ELEM (i + 1, l);
t1[l - ll + 2 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i + 1, l + 1);
}
if (uisec < isec)
{
t1[l - ll + 1 + (isec << 8) - 257] =
- a[ii + isec - 1 + l * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l);
t1[l - ll + 2 + (isec << 8) - 257] =
- a[ii + isec - 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l + 1);
}
}
if (ulsec < lsec)
@@ -2113,7 +2176,7 @@ matmul_c10_vanilla (gfc_array_c10 * const restrict retarray,
for (i = ii; i< i4; ++i)
{
t1[lsec + ((i - ii + 1) << 8) - 257] =
- a[i + (ll + lsec - 1) * a_dim1];
+ A_ARRAY_ELEM (i, ll + lsec - 1);
}
}
@@ -2124,100 +2187,100 @@ matmul_c10_vanilla (gfc_array_c10 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f22 = c[i + 1 + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f23 = c[i + 1 + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
- f24 = c[i + 1 + (j + 3) * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
- f32 = c[i + 2 + (j + 1) * c_dim1];
- f42 = c[i + 3 + (j + 1) * c_dim1];
- f33 = c[i + 2 + (j + 2) * c_dim1];
- f43 = c[i + 3 + (j + 2) * c_dim1];
- f34 = c[i + 2 + (j + 3) * c_dim1];
- f44 = c[i + 3 + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f22 = C_ARRAY_ELEM (i + 1, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f23 = C_ARRAY_ELEM (i + 1, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
+ f24 = C_ARRAY_ELEM (i + 1, j + 3);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
+ f32 = C_ARRAY_ELEM (i + 2, j + 1);
+ f42 = C_ARRAY_ELEM (i + 3, j + 1);
+ f33 = C_ARRAY_ELEM (i + 2, j + 2);
+ f43 = C_ARRAY_ELEM (i + 3, j + 2);
+ f34 = C_ARRAY_ELEM (i + 2, j + 3);
+ f44 = C_ARRAY_ELEM (i + 3, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f22 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f23 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f24 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f32 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f42 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f33 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f43 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f34 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f44 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + 1 + (j + 1) * c_dim1] = f22;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + 1 + (j + 2) * c_dim1] = f23;
- c[i + (j + 3) * c_dim1] = f14;
- c[i + 1 + (j + 3) * c_dim1] = f24;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
- c[i + 2 + (j + 1) * c_dim1] = f32;
- c[i + 3 + (j + 1) * c_dim1] = f42;
- c[i + 2 + (j + 2) * c_dim1] = f33;
- c[i + 3 + (j + 2) * c_dim1] = f43;
- c[i + 2 + (j + 3) * c_dim1] = f34;
- c[i + 3 + (j + 3) * c_dim1] = f44;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i + 1, j + 1) = f22;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i + 1, j + 2) = f23;
+ C_ARRAY_ELEM (i, j + 3) = f14;
+ C_ARRAY_ELEM (i + 1, j + 3) = f24;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
+ C_ARRAY_ELEM (i + 2, j + 1) = f32;
+ C_ARRAY_ELEM (i + 3, j + 1) = f42;
+ C_ARRAY_ELEM (i + 2, j + 2) = f33;
+ C_ARRAY_ELEM (i + 3, j + 2) = f43;
+ C_ARRAY_ELEM (i + 2, j + 3) = f34;
+ C_ARRAY_ELEM (i + 3, j + 3) = f44;
}
if (uisec < isec)
{
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 1) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 2) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 3) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + (j + 3) * c_dim1] = f14;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i, j + 3) = f14;
}
}
}
@@ -2229,38 +2292,38 @@ matmul_c10_vanilla (gfc_array_c10 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
}
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
+ C_ARRAY_ELEM (i, j) = f11;
}
}
}
@@ -2269,6 +2332,9 @@ matmul_c10_vanilla (gfc_array_c10 * const restrict retarray,
}
free(t1);
return;
+#undef A_ARRAY_ELEM
+#undef B_ARRAY_ELEM
+#undef C_ARRAY_ELEM
}
else if (rxstride == 1 && aystride == 1 && bxstride == 1)
{
@@ -2281,11 +2347,11 @@ matmul_c10_vanilla (gfc_array_c10 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_10) 0;
for (n = 0; n < count; n++)
s += abase_x[n] * bbase_y[n];
@@ -2300,11 +2366,11 @@ matmul_c10_vanilla (gfc_array_c10 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_10) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n];
- dest[y*rystride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n) * bbase_y[n];
+ ARRAY_ELEM_AT_OFFSET (dest, y * rystride_bytes) = s;
}
}
}
@@ -2315,26 +2381,27 @@ matmul_c10_vanilla (gfc_array_c10 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_10) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n*bxstride];
- dest[y*rxstride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
}
}
else if (axstride < aystride)
{
for (y = 0; y < ycount; y++)
for (x = 0; x < xcount; x++)
- dest[x*rxstride + y*rystride] = (GFC_COMPLEX_10)0;
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y) = (GFC_COMPLEX_10)0;
for (y = 0; y < ycount; y++)
for (n = 0; n < count; n++)
for (x = 0; x < xcount; x++)
/* dest[x,y] += a[x,n] * b[n,y] */
- dest[x*rxstride + y*rystride] +=
- abase[x*axstride + n*aystride] *
- bbase[n*bxstride + y*bystride];
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y)
+ += GFC_DESCRIPTOR2_ELEM (a, x, n)
+ * GFC_DESCRIPTOR2_ELEM (b, n, y);
}
else
{
@@ -2345,15 +2412,16 @@ matmul_c10_vanilla (gfc_array_c10 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_10) 0;
for (n = 0; n < count; n++)
- s += abase_x[n*aystride] * bbase_y[n*bxstride];
- dest_y[x*rxstride] = s;
+ s += ARRAY_ELEM_AT_OFFSET (abase_x, n * aystride_bytes)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ ARRAY_ELEM_AT_OFFSET (dest_y, x * rxstride_bytes) = s;
}
}
}
@@ -2453,7 +2521,8 @@ matmul_c10 (gfc_array_c10 * const restrict retarray,
index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
index_type x, y, n, count, xcount, ycount;
- index_type aystride_bytes, bystride_bytes, rystride_bytes;
+ index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
+ rxstride_bytes, rystride_bytes;
assert (GFC_DESCRIPTOR_RANK (a) == 2
|| GFC_DESCRIPTOR_RANK (b) == 2);
@@ -2542,12 +2611,13 @@ matmul_c10 (gfc_array_c10 * const restrict retarray,
either as a row or a column matrix. We want both cases to
work. */
rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
+ rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
}
else
{
rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
+ rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
}
@@ -2556,6 +2626,7 @@ matmul_c10 (gfc_array_c10 * const restrict retarray,
/* Treat it as a a row matrix A[1,count]. */
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = 1;
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = sizeof (GFC_COMPLEX_10);
xcount = 1;
@@ -2565,6 +2636,7 @@ matmul_c10 (gfc_array_c10 * const restrict retarray,
{
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = GFC_DESCRIPTOR_STRIDE(a,1);
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,1);
count = GFC_DESCRIPTOR_EXTENT(a,1);
@@ -2583,17 +2655,20 @@ matmul_c10 (gfc_array_c10 * const restrict retarray,
{
/* Treat it as a column matrix B[count,1] */
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
/* bystride should never be used for 1-dimensional b.
The value is only used for calculation of the
memory by the buffer. */
bystride = 256;
+ bystride_bytes = 99999999;
ycount = 1;
}
else
{
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
bystride = GFC_DESCRIPTOR_STRIDE(b,1);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
bystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,1);
ycount = GFC_DESCRIPTOR_EXTENT(b,1);
}
@@ -2653,12 +2728,11 @@ matmul_c10 (gfc_array_c10 * const restrict retarray,
from netlib.org, translated to C, and modified for matmul.m4. */
- const GFC_COMPLEX_10 *a, *b;
GFC_COMPLEX_10 *c;
const index_type m = xcount, n = ycount, k = count;
/* System generated locals */
- index_type a_dim1, b_dim1, c_dim1,
+ index_type a_dim1, b_dim1,
i1, i2, i3, i4, i5, i6;
/* Local variables */
@@ -2668,19 +2742,25 @@ matmul_c10 (gfc_array_c10 * const restrict retarray,
index_type isec, jsec, lsec, uisec, ujsec, ulsec;
GFC_COMPLEX_10 *t1;
- a = abase;
- b = bbase;
c = retarray->base_addr;
/* Parameter adjustments */
- c_dim1 = rystride;
a_dim1 = aystride;
b_dim1 = bystride;
- /* Empty c first. */
+#define A_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (abase, (i) * sizeof (GFC_COMPLEX_10) + (j) * aystride_bytes))
+
+#define B_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (bbase, (i) * sizeof (GFC_COMPLEX_10) + (j) * bystride_bytes))
+
+#define C_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (c, (i) * sizeof (GFC_COMPLEX_10) + (j) * rystride_bytes))
+
+ /* Empty result first. */
for (j=0; j<n; j++)
for (i=0; i<m; i++)
- c[i + j * c_dim1] = (GFC_COMPLEX_10)0;
+ C_ARRAY_ELEM (i, j) = (GFC_COMPLEX_10)0;
/* Early exit if possible */
if (m == 0 || n == 0 || k == 0)
@@ -2732,20 +2812,20 @@ matmul_c10 (gfc_array_c10 * const restrict retarray,
for (i = ii; i < i5; i += 2)
{
t1[l - ll + 1 + ((i - ii + 1) << 8) - 257] =
- a[i + l * a_dim1];
+ A_ARRAY_ELEM (i, l);
t1[l - ll + 2 + ((i - ii + 1) << 8) - 257] =
- a[i + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i, l + 1);
t1[l - ll + 1 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + l * a_dim1];
+ A_ARRAY_ELEM (i + 1, l);
t1[l - ll + 2 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i + 1, l + 1);
}
if (uisec < isec)
{
t1[l - ll + 1 + (isec << 8) - 257] =
- a[ii + isec - 1 + l * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l);
t1[l - ll + 2 + (isec << 8) - 257] =
- a[ii + isec - 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l + 1);
}
}
if (ulsec < lsec)
@@ -2754,7 +2834,7 @@ matmul_c10 (gfc_array_c10 * const restrict retarray,
for (i = ii; i< i4; ++i)
{
t1[lsec + ((i - ii + 1) << 8) - 257] =
- a[i + (ll + lsec - 1) * a_dim1];
+ A_ARRAY_ELEM (i, ll + lsec - 1);
}
}
@@ -2765,100 +2845,100 @@ matmul_c10 (gfc_array_c10 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f22 = c[i + 1 + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f23 = c[i + 1 + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
- f24 = c[i + 1 + (j + 3) * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
- f32 = c[i + 2 + (j + 1) * c_dim1];
- f42 = c[i + 3 + (j + 1) * c_dim1];
- f33 = c[i + 2 + (j + 2) * c_dim1];
- f43 = c[i + 3 + (j + 2) * c_dim1];
- f34 = c[i + 2 + (j + 3) * c_dim1];
- f44 = c[i + 3 + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f22 = C_ARRAY_ELEM (i + 1, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f23 = C_ARRAY_ELEM (i + 1, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
+ f24 = C_ARRAY_ELEM (i + 1, j + 3);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
+ f32 = C_ARRAY_ELEM (i + 2, j + 1);
+ f42 = C_ARRAY_ELEM (i + 3, j + 1);
+ f33 = C_ARRAY_ELEM (i + 2, j + 2);
+ f43 = C_ARRAY_ELEM (i + 3, j + 2);
+ f34 = C_ARRAY_ELEM (i + 2, j + 3);
+ f44 = C_ARRAY_ELEM (i + 3, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f22 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f23 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f24 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f32 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f42 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f33 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f43 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f34 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f44 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + 1 + (j + 1) * c_dim1] = f22;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + 1 + (j + 2) * c_dim1] = f23;
- c[i + (j + 3) * c_dim1] = f14;
- c[i + 1 + (j + 3) * c_dim1] = f24;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
- c[i + 2 + (j + 1) * c_dim1] = f32;
- c[i + 3 + (j + 1) * c_dim1] = f42;
- c[i + 2 + (j + 2) * c_dim1] = f33;
- c[i + 3 + (j + 2) * c_dim1] = f43;
- c[i + 2 + (j + 3) * c_dim1] = f34;
- c[i + 3 + (j + 3) * c_dim1] = f44;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i + 1, j + 1) = f22;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i + 1, j + 2) = f23;
+ C_ARRAY_ELEM (i, j + 3) = f14;
+ C_ARRAY_ELEM (i + 1, j + 3) = f24;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
+ C_ARRAY_ELEM (i + 2, j + 1) = f32;
+ C_ARRAY_ELEM (i + 3, j + 1) = f42;
+ C_ARRAY_ELEM (i + 2, j + 2) = f33;
+ C_ARRAY_ELEM (i + 3, j + 2) = f43;
+ C_ARRAY_ELEM (i + 2, j + 3) = f34;
+ C_ARRAY_ELEM (i + 3, j + 3) = f44;
}
if (uisec < isec)
{
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 1) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 2) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 3) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + (j + 3) * c_dim1] = f14;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i, j + 3) = f14;
}
}
}
@@ -2870,38 +2950,38 @@ matmul_c10 (gfc_array_c10 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
}
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
+ C_ARRAY_ELEM (i, j) = f11;
}
}
}
@@ -2910,6 +2990,9 @@ matmul_c10 (gfc_array_c10 * const restrict retarray,
}
free(t1);
return;
+#undef A_ARRAY_ELEM
+#undef B_ARRAY_ELEM
+#undef C_ARRAY_ELEM
}
else if (rxstride == 1 && aystride == 1 && bxstride == 1)
{
@@ -2922,11 +3005,11 @@ matmul_c10 (gfc_array_c10 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_10) 0;
for (n = 0; n < count; n++)
s += abase_x[n] * bbase_y[n];
@@ -2941,11 +3024,11 @@ matmul_c10 (gfc_array_c10 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_10) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n];
- dest[y*rystride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n) * bbase_y[n];
+ ARRAY_ELEM_AT_OFFSET (dest, y * rystride_bytes) = s;
}
}
}
@@ -2956,26 +3039,27 @@ matmul_c10 (gfc_array_c10 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_10) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n*bxstride];
- dest[y*rxstride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
}
}
else if (axstride < aystride)
{
for (y = 0; y < ycount; y++)
for (x = 0; x < xcount; x++)
- dest[x*rxstride + y*rystride] = (GFC_COMPLEX_10)0;
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y) = (GFC_COMPLEX_10)0;
for (y = 0; y < ycount; y++)
for (n = 0; n < count; n++)
for (x = 0; x < xcount; x++)
/* dest[x,y] += a[x,n] * b[n,y] */
- dest[x*rxstride + y*rystride] +=
- abase[x*axstride + n*aystride] *
- bbase[n*bxstride + y*bystride];
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y)
+ += GFC_DESCRIPTOR2_ELEM (a, x, n)
+ * GFC_DESCRIPTOR2_ELEM (b, n, y);
}
else
{
@@ -2986,15 +3070,16 @@ matmul_c10 (gfc_array_c10 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_10) 0;
for (n = 0; n < count; n++)
- s += abase_x[n*aystride] * bbase_y[n*bxstride];
- dest_y[x*rxstride] = s;
+ s += ARRAY_ELEM_AT_OFFSET (abase_x, n * aystride_bytes)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ ARRAY_ELEM_AT_OFFSET (dest_y, x * rxstride_bytes) = s;
}
}
}
diff --git a/libgfortran/generated/matmul_c16.c b/libgfortran/generated/matmul_c16.c
index 08c80be605db..8d592540b553 100644
--- a/libgfortran/generated/matmul_c16.c
+++ b/libgfortran/generated/matmul_c16.c
@@ -94,7 +94,8 @@ matmul_c16_avx (gfc_array_c16 * const restrict retarray,
index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
index_type x, y, n, count, xcount, ycount;
- index_type aystride_bytes, bystride_bytes, rystride_bytes;
+ index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
+ rxstride_bytes, rystride_bytes;
assert (GFC_DESCRIPTOR_RANK (a) == 2
|| GFC_DESCRIPTOR_RANK (b) == 2);
@@ -183,12 +184,13 @@ matmul_c16_avx (gfc_array_c16 * const restrict retarray,
either as a row or a column matrix. We want both cases to
work. */
rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
+ rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
}
else
{
rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
+ rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
}
@@ -197,6 +199,7 @@ matmul_c16_avx (gfc_array_c16 * const restrict retarray,
/* Treat it as a a row matrix A[1,count]. */
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = 1;
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = sizeof (GFC_COMPLEX_16);
xcount = 1;
@@ -206,6 +209,7 @@ matmul_c16_avx (gfc_array_c16 * const restrict retarray,
{
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = GFC_DESCRIPTOR_STRIDE(a,1);
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,1);
count = GFC_DESCRIPTOR_EXTENT(a,1);
@@ -224,17 +228,20 @@ matmul_c16_avx (gfc_array_c16 * const restrict retarray,
{
/* Treat it as a column matrix B[count,1] */
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
/* bystride should never be used for 1-dimensional b.
The value is only used for calculation of the
memory by the buffer. */
bystride = 256;
+ bystride_bytes = 99999999;
ycount = 1;
}
else
{
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
bystride = GFC_DESCRIPTOR_STRIDE(b,1);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
bystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,1);
ycount = GFC_DESCRIPTOR_EXTENT(b,1);
}
@@ -294,12 +301,11 @@ matmul_c16_avx (gfc_array_c16 * const restrict retarray,
from netlib.org, translated to C, and modified for matmul.m4. */
- const GFC_COMPLEX_16 *a, *b;
GFC_COMPLEX_16 *c;
const index_type m = xcount, n = ycount, k = count;
/* System generated locals */
- index_type a_dim1, b_dim1, c_dim1,
+ index_type a_dim1, b_dim1,
i1, i2, i3, i4, i5, i6;
/* Local variables */
@@ -309,19 +315,25 @@ matmul_c16_avx (gfc_array_c16 * const restrict retarray,
index_type isec, jsec, lsec, uisec, ujsec, ulsec;
GFC_COMPLEX_16 *t1;
- a = abase;
- b = bbase;
c = retarray->base_addr;
/* Parameter adjustments */
- c_dim1 = rystride;
a_dim1 = aystride;
b_dim1 = bystride;
- /* Empty c first. */
+#define A_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (abase, (i) * sizeof (GFC_COMPLEX_16) + (j) * aystride_bytes))
+
+#define B_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (bbase, (i) * sizeof (GFC_COMPLEX_16) + (j) * bystride_bytes))
+
+#define C_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (c, (i) * sizeof (GFC_COMPLEX_16) + (j) * rystride_bytes))
+
+ /* Empty result first. */
for (j=0; j<n; j++)
for (i=0; i<m; i++)
- c[i + j * c_dim1] = (GFC_COMPLEX_16)0;
+ C_ARRAY_ELEM (i, j) = (GFC_COMPLEX_16)0;
/* Early exit if possible */
if (m == 0 || n == 0 || k == 0)
@@ -373,20 +385,20 @@ matmul_c16_avx (gfc_array_c16 * const restrict retarray,
for (i = ii; i < i5; i += 2)
{
t1[l - ll + 1 + ((i - ii + 1) << 8) - 257] =
- a[i + l * a_dim1];
+ A_ARRAY_ELEM (i, l);
t1[l - ll + 2 + ((i - ii + 1) << 8) - 257] =
- a[i + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i, l + 1);
t1[l - ll + 1 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + l * a_dim1];
+ A_ARRAY_ELEM (i + 1, l);
t1[l - ll + 2 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i + 1, l + 1);
}
if (uisec < isec)
{
t1[l - ll + 1 + (isec << 8) - 257] =
- a[ii + isec - 1 + l * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l);
t1[l - ll + 2 + (isec << 8) - 257] =
- a[ii + isec - 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l + 1);
}
}
if (ulsec < lsec)
@@ -395,7 +407,7 @@ matmul_c16_avx (gfc_array_c16 * const restrict retarray,
for (i = ii; i< i4; ++i)
{
t1[lsec + ((i - ii + 1) << 8) - 257] =
- a[i + (ll + lsec - 1) * a_dim1];
+ A_ARRAY_ELEM (i, ll + lsec - 1);
}
}
@@ -406,100 +418,100 @@ matmul_c16_avx (gfc_array_c16 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f22 = c[i + 1 + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f23 = c[i + 1 + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
- f24 = c[i + 1 + (j + 3) * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
- f32 = c[i + 2 + (j + 1) * c_dim1];
- f42 = c[i + 3 + (j + 1) * c_dim1];
- f33 = c[i + 2 + (j + 2) * c_dim1];
- f43 = c[i + 3 + (j + 2) * c_dim1];
- f34 = c[i + 2 + (j + 3) * c_dim1];
- f44 = c[i + 3 + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f22 = C_ARRAY_ELEM (i + 1, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f23 = C_ARRAY_ELEM (i + 1, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
+ f24 = C_ARRAY_ELEM (i + 1, j + 3);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
+ f32 = C_ARRAY_ELEM (i + 2, j + 1);
+ f42 = C_ARRAY_ELEM (i + 3, j + 1);
+ f33 = C_ARRAY_ELEM (i + 2, j + 2);
+ f43 = C_ARRAY_ELEM (i + 3, j + 2);
+ f34 = C_ARRAY_ELEM (i + 2, j + 3);
+ f44 = C_ARRAY_ELEM (i + 3, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f22 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f23 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f24 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f32 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f42 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f33 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f43 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f34 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f44 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + 1 + (j + 1) * c_dim1] = f22;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + 1 + (j + 2) * c_dim1] = f23;
- c[i + (j + 3) * c_dim1] = f14;
- c[i + 1 + (j + 3) * c_dim1] = f24;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
- c[i + 2 + (j + 1) * c_dim1] = f32;
- c[i + 3 + (j + 1) * c_dim1] = f42;
- c[i + 2 + (j + 2) * c_dim1] = f33;
- c[i + 3 + (j + 2) * c_dim1] = f43;
- c[i + 2 + (j + 3) * c_dim1] = f34;
- c[i + 3 + (j + 3) * c_dim1] = f44;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i + 1, j + 1) = f22;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i + 1, j + 2) = f23;
+ C_ARRAY_ELEM (i, j + 3) = f14;
+ C_ARRAY_ELEM (i + 1, j + 3) = f24;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
+ C_ARRAY_ELEM (i + 2, j + 1) = f32;
+ C_ARRAY_ELEM (i + 3, j + 1) = f42;
+ C_ARRAY_ELEM (i + 2, j + 2) = f33;
+ C_ARRAY_ELEM (i + 3, j + 2) = f43;
+ C_ARRAY_ELEM (i + 2, j + 3) = f34;
+ C_ARRAY_ELEM (i + 3, j + 3) = f44;
}
if (uisec < isec)
{
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 1) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 2) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 3) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + (j + 3) * c_dim1] = f14;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i, j + 3) = f14;
}
}
}
@@ -511,38 +523,38 @@ matmul_c16_avx (gfc_array_c16 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
}
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
+ C_ARRAY_ELEM (i, j) = f11;
}
}
}
@@ -551,6 +563,9 @@ matmul_c16_avx (gfc_array_c16 * const restrict retarray,
}
free(t1);
return;
+#undef A_ARRAY_ELEM
+#undef B_ARRAY_ELEM
+#undef C_ARRAY_ELEM
}
else if (rxstride == 1 && aystride == 1 && bxstride == 1)
{
@@ -563,11 +578,11 @@ matmul_c16_avx (gfc_array_c16 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_16) 0;
for (n = 0; n < count; n++)
s += abase_x[n] * bbase_y[n];
@@ -582,11 +597,11 @@ matmul_c16_avx (gfc_array_c16 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_16) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n];
- dest[y*rystride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n) * bbase_y[n];
+ ARRAY_ELEM_AT_OFFSET (dest, y * rystride_bytes) = s;
}
}
}
@@ -597,26 +612,27 @@ matmul_c16_avx (gfc_array_c16 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_16) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n*bxstride];
- dest[y*rxstride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
}
}
else if (axstride < aystride)
{
for (y = 0; y < ycount; y++)
for (x = 0; x < xcount; x++)
- dest[x*rxstride + y*rystride] = (GFC_COMPLEX_16)0;
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y) = (GFC_COMPLEX_16)0;
for (y = 0; y < ycount; y++)
for (n = 0; n < count; n++)
for (x = 0; x < xcount; x++)
/* dest[x,y] += a[x,n] * b[n,y] */
- dest[x*rxstride + y*rystride] +=
- abase[x*axstride + n*aystride] *
- bbase[n*bxstride + y*bystride];
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y)
+ += GFC_DESCRIPTOR2_ELEM (a, x, n)
+ * GFC_DESCRIPTOR2_ELEM (b, n, y);
}
else
{
@@ -627,15 +643,16 @@ matmul_c16_avx (gfc_array_c16 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_16) 0;
for (n = 0; n < count; n++)
- s += abase_x[n*aystride] * bbase_y[n*bxstride];
- dest_y[x*rxstride] = s;
+ s += ARRAY_ELEM_AT_OFFSET (abase_x, n * aystride_bytes)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ ARRAY_ELEM_AT_OFFSET (dest_y, x * rxstride_bytes) = s;
}
}
}
@@ -662,7 +679,8 @@ matmul_c16_avx2 (gfc_array_c16 * const restrict retarray,
index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
index_type x, y, n, count, xcount, ycount;
- index_type aystride_bytes, bystride_bytes, rystride_bytes;
+ index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
+ rxstride_bytes, rystride_bytes;
assert (GFC_DESCRIPTOR_RANK (a) == 2
|| GFC_DESCRIPTOR_RANK (b) == 2);
@@ -751,12 +769,13 @@ matmul_c16_avx2 (gfc_array_c16 * const restrict retarray,
either as a row or a column matrix. We want both cases to
work. */
rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
+ rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
}
else
{
rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
+ rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
}
@@ -765,6 +784,7 @@ matmul_c16_avx2 (gfc_array_c16 * const restrict retarray,
/* Treat it as a a row matrix A[1,count]. */
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = 1;
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = sizeof (GFC_COMPLEX_16);
xcount = 1;
@@ -774,6 +794,7 @@ matmul_c16_avx2 (gfc_array_c16 * const restrict retarray,
{
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = GFC_DESCRIPTOR_STRIDE(a,1);
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,1);
count = GFC_DESCRIPTOR_EXTENT(a,1);
@@ -792,17 +813,20 @@ matmul_c16_avx2 (gfc_array_c16 * const restrict retarray,
{
/* Treat it as a column matrix B[count,1] */
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
/* bystride should never be used for 1-dimensional b.
The value is only used for calculation of the
memory by the buffer. */
bystride = 256;
+ bystride_bytes = 99999999;
ycount = 1;
}
else
{
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
bystride = GFC_DESCRIPTOR_STRIDE(b,1);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
bystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,1);
ycount = GFC_DESCRIPTOR_EXTENT(b,1);
}
@@ -862,12 +886,11 @@ matmul_c16_avx2 (gfc_array_c16 * const restrict retarray,
from netlib.org, translated to C, and modified for matmul.m4. */
- const GFC_COMPLEX_16 *a, *b;
GFC_COMPLEX_16 *c;
const index_type m = xcount, n = ycount, k = count;
/* System generated locals */
- index_type a_dim1, b_dim1, c_dim1,
+ index_type a_dim1, b_dim1,
i1, i2, i3, i4, i5, i6;
/* Local variables */
@@ -877,19 +900,25 @@ matmul_c16_avx2 (gfc_array_c16 * const restrict retarray,
index_type isec, jsec, lsec, uisec, ujsec, ulsec;
GFC_COMPLEX_16 *t1;
- a = abase;
- b = bbase;
c = retarray->base_addr;
/* Parameter adjustments */
- c_dim1 = rystride;
a_dim1 = aystride;
b_dim1 = bystride;
- /* Empty c first. */
+#define A_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (abase, (i) * sizeof (GFC_COMPLEX_16) + (j) * aystride_bytes))
+
+#define B_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (bbase, (i) * sizeof (GFC_COMPLEX_16) + (j) * bystride_bytes))
+
+#define C_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (c, (i) * sizeof (GFC_COMPLEX_16) + (j) * rystride_bytes))
+
+ /* Empty result first. */
for (j=0; j<n; j++)
for (i=0; i<m; i++)
- c[i + j * c_dim1] = (GFC_COMPLEX_16)0;
+ C_ARRAY_ELEM (i, j) = (GFC_COMPLEX_16)0;
/* Early exit if possible */
if (m == 0 || n == 0 || k == 0)
@@ -941,20 +970,20 @@ matmul_c16_avx2 (gfc_array_c16 * const restrict retarray,
for (i = ii; i < i5; i += 2)
{
t1[l - ll + 1 + ((i - ii + 1) << 8) - 257] =
- a[i + l * a_dim1];
+ A_ARRAY_ELEM (i, l);
t1[l - ll + 2 + ((i - ii + 1) << 8) - 257] =
- a[i + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i, l + 1);
t1[l - ll + 1 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + l * a_dim1];
+ A_ARRAY_ELEM (i + 1, l);
t1[l - ll + 2 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i + 1, l + 1);
}
if (uisec < isec)
{
t1[l - ll + 1 + (isec << 8) - 257] =
- a[ii + isec - 1 + l * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l);
t1[l - ll + 2 + (isec << 8) - 257] =
- a[ii + isec - 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l + 1);
}
}
if (ulsec < lsec)
@@ -963,7 +992,7 @@ matmul_c16_avx2 (gfc_array_c16 * const restrict retarray,
for (i = ii; i< i4; ++i)
{
t1[lsec + ((i - ii + 1) << 8) - 257] =
- a[i + (ll + lsec - 1) * a_dim1];
+ A_ARRAY_ELEM (i, ll + lsec - 1);
}
}
@@ -974,100 +1003,100 @@ matmul_c16_avx2 (gfc_array_c16 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f22 = c[i + 1 + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f23 = c[i + 1 + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
- f24 = c[i + 1 + (j + 3) * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
- f32 = c[i + 2 + (j + 1) * c_dim1];
- f42 = c[i + 3 + (j + 1) * c_dim1];
- f33 = c[i + 2 + (j + 2) * c_dim1];
- f43 = c[i + 3 + (j + 2) * c_dim1];
- f34 = c[i + 2 + (j + 3) * c_dim1];
- f44 = c[i + 3 + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f22 = C_ARRAY_ELEM (i + 1, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f23 = C_ARRAY_ELEM (i + 1, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
+ f24 = C_ARRAY_ELEM (i + 1, j + 3);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
+ f32 = C_ARRAY_ELEM (i + 2, j + 1);
+ f42 = C_ARRAY_ELEM (i + 3, j + 1);
+ f33 = C_ARRAY_ELEM (i + 2, j + 2);
+ f43 = C_ARRAY_ELEM (i + 3, j + 2);
+ f34 = C_ARRAY_ELEM (i + 2, j + 3);
+ f44 = C_ARRAY_ELEM (i + 3, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f22 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f23 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f24 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f32 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f42 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f33 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f43 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f34 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f44 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + 1 + (j + 1) * c_dim1] = f22;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + 1 + (j + 2) * c_dim1] = f23;
- c[i + (j + 3) * c_dim1] = f14;
- c[i + 1 + (j + 3) * c_dim1] = f24;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
- c[i + 2 + (j + 1) * c_dim1] = f32;
- c[i + 3 + (j + 1) * c_dim1] = f42;
- c[i + 2 + (j + 2) * c_dim1] = f33;
- c[i + 3 + (j + 2) * c_dim1] = f43;
- c[i + 2 + (j + 3) * c_dim1] = f34;
- c[i + 3 + (j + 3) * c_dim1] = f44;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i + 1, j + 1) = f22;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i + 1, j + 2) = f23;
+ C_ARRAY_ELEM (i, j + 3) = f14;
+ C_ARRAY_ELEM (i + 1, j + 3) = f24;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
+ C_ARRAY_ELEM (i + 2, j + 1) = f32;
+ C_ARRAY_ELEM (i + 3, j + 1) = f42;
+ C_ARRAY_ELEM (i + 2, j + 2) = f33;
+ C_ARRAY_ELEM (i + 3, j + 2) = f43;
+ C_ARRAY_ELEM (i + 2, j + 3) = f34;
+ C_ARRAY_ELEM (i + 3, j + 3) = f44;
}
if (uisec < isec)
{
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 1) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 2) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 3) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + (j + 3) * c_dim1] = f14;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i, j + 3) = f14;
}
}
}
@@ -1079,38 +1108,38 @@ matmul_c16_avx2 (gfc_array_c16 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
}
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
+ C_ARRAY_ELEM (i, j) = f11;
}
}
}
@@ -1119,6 +1148,9 @@ matmul_c16_avx2 (gfc_array_c16 * const restrict retarray,
}
free(t1);
return;
+#undef A_ARRAY_ELEM
+#undef B_ARRAY_ELEM
+#undef C_ARRAY_ELEM
}
else if (rxstride == 1 && aystride == 1 && bxstride == 1)
{
@@ -1131,11 +1163,11 @@ matmul_c16_avx2 (gfc_array_c16 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_16) 0;
for (n = 0; n < count; n++)
s += abase_x[n] * bbase_y[n];
@@ -1150,11 +1182,11 @@ matmul_c16_avx2 (gfc_array_c16 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_16) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n];
- dest[y*rystride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n) * bbase_y[n];
+ ARRAY_ELEM_AT_OFFSET (dest, y * rystride_bytes) = s;
}
}
}
@@ -1165,26 +1197,27 @@ matmul_c16_avx2 (gfc_array_c16 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_16) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n*bxstride];
- dest[y*rxstride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
}
}
else if (axstride < aystride)
{
for (y = 0; y < ycount; y++)
for (x = 0; x < xcount; x++)
- dest[x*rxstride + y*rystride] = (GFC_COMPLEX_16)0;
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y) = (GFC_COMPLEX_16)0;
for (y = 0; y < ycount; y++)
for (n = 0; n < count; n++)
for (x = 0; x < xcount; x++)
/* dest[x,y] += a[x,n] * b[n,y] */
- dest[x*rxstride + y*rystride] +=
- abase[x*axstride + n*aystride] *
- bbase[n*bxstride + y*bystride];
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y)
+ += GFC_DESCRIPTOR2_ELEM (a, x, n)
+ * GFC_DESCRIPTOR2_ELEM (b, n, y);
}
else
{
@@ -1195,15 +1228,16 @@ matmul_c16_avx2 (gfc_array_c16 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_16) 0;
for (n = 0; n < count; n++)
- s += abase_x[n*aystride] * bbase_y[n*bxstride];
- dest_y[x*rxstride] = s;
+ s += ARRAY_ELEM_AT_OFFSET (abase_x, n * aystride_bytes)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ ARRAY_ELEM_AT_OFFSET (dest_y, x * rxstride_bytes) = s;
}
}
}
@@ -1230,7 +1264,8 @@ matmul_c16_avx512f (gfc_array_c16 * const restrict retarray,
index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
index_type x, y, n, count, xcount, ycount;
- index_type aystride_bytes, bystride_bytes, rystride_bytes;
+ index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
+ rxstride_bytes, rystride_bytes;
assert (GFC_DESCRIPTOR_RANK (a) == 2
|| GFC_DESCRIPTOR_RANK (b) == 2);
@@ -1319,12 +1354,13 @@ matmul_c16_avx512f (gfc_array_c16 * const restrict retarray,
either as a row or a column matrix. We want both cases to
work. */
rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
+ rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
}
else
{
rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
+ rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
}
@@ -1333,6 +1369,7 @@ matmul_c16_avx512f (gfc_array_c16 * const restrict retarray,
/* Treat it as a a row matrix A[1,count]. */
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = 1;
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = sizeof (GFC_COMPLEX_16);
xcount = 1;
@@ -1342,6 +1379,7 @@ matmul_c16_avx512f (gfc_array_c16 * const restrict retarray,
{
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = GFC_DESCRIPTOR_STRIDE(a,1);
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,1);
count = GFC_DESCRIPTOR_EXTENT(a,1);
@@ -1360,17 +1398,20 @@ matmul_c16_avx512f (gfc_array_c16 * const restrict retarray,
{
/* Treat it as a column matrix B[count,1] */
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
/* bystride should never be used for 1-dimensional b.
The value is only used for calculation of the
memory by the buffer. */
bystride = 256;
+ bystride_bytes = 99999999;
ycount = 1;
}
else
{
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
bystride = GFC_DESCRIPTOR_STRIDE(b,1);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
bystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,1);
ycount = GFC_DESCRIPTOR_EXTENT(b,1);
}
@@ -1430,12 +1471,11 @@ matmul_c16_avx512f (gfc_array_c16 * const restrict retarray,
from netlib.org, translated to C, and modified for matmul.m4. */
- const GFC_COMPLEX_16 *a, *b;
GFC_COMPLEX_16 *c;
const index_type m = xcount, n = ycount, k = count;
/* System generated locals */
- index_type a_dim1, b_dim1, c_dim1,
+ index_type a_dim1, b_dim1,
i1, i2, i3, i4, i5, i6;
/* Local variables */
@@ -1445,19 +1485,25 @@ matmul_c16_avx512f (gfc_array_c16 * const restrict retarray,
index_type isec, jsec, lsec, uisec, ujsec, ulsec;
GFC_COMPLEX_16 *t1;
- a = abase;
- b = bbase;
c = retarray->base_addr;
/* Parameter adjustments */
- c_dim1 = rystride;
a_dim1 = aystride;
b_dim1 = bystride;
- /* Empty c first. */
+#define A_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (abase, (i) * sizeof (GFC_COMPLEX_16) + (j) * aystride_bytes))
+
+#define B_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (bbase, (i) * sizeof (GFC_COMPLEX_16) + (j) * bystride_bytes))
+
+#define C_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (c, (i) * sizeof (GFC_COMPLEX_16) + (j) * rystride_bytes))
+
+ /* Empty result first. */
for (j=0; j<n; j++)
for (i=0; i<m; i++)
- c[i + j * c_dim1] = (GFC_COMPLEX_16)0;
+ C_ARRAY_ELEM (i, j) = (GFC_COMPLEX_16)0;
/* Early exit if possible */
if (m == 0 || n == 0 || k == 0)
@@ -1509,20 +1555,20 @@ matmul_c16_avx512f (gfc_array_c16 * const restrict retarray,
for (i = ii; i < i5; i += 2)
{
t1[l - ll + 1 + ((i - ii + 1) << 8) - 257] =
- a[i + l * a_dim1];
+ A_ARRAY_ELEM (i, l);
t1[l - ll + 2 + ((i - ii + 1) << 8) - 257] =
- a[i + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i, l + 1);
t1[l - ll + 1 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + l * a_dim1];
+ A_ARRAY_ELEM (i + 1, l);
t1[l - ll + 2 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i + 1, l + 1);
}
if (uisec < isec)
{
t1[l - ll + 1 + (isec << 8) - 257] =
- a[ii + isec - 1 + l * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l);
t1[l - ll + 2 + (isec << 8) - 257] =
- a[ii + isec - 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l + 1);
}
}
if (ulsec < lsec)
@@ -1531,7 +1577,7 @@ matmul_c16_avx512f (gfc_array_c16 * const restrict retarray,
for (i = ii; i< i4; ++i)
{
t1[lsec + ((i - ii + 1) << 8) - 257] =
- a[i + (ll + lsec - 1) * a_dim1];
+ A_ARRAY_ELEM (i, ll + lsec - 1);
}
}
@@ -1542,100 +1588,100 @@ matmul_c16_avx512f (gfc_array_c16 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f22 = c[i + 1 + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f23 = c[i + 1 + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
- f24 = c[i + 1 + (j + 3) * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
- f32 = c[i + 2 + (j + 1) * c_dim1];
- f42 = c[i + 3 + (j + 1) * c_dim1];
- f33 = c[i + 2 + (j + 2) * c_dim1];
- f43 = c[i + 3 + (j + 2) * c_dim1];
- f34 = c[i + 2 + (j + 3) * c_dim1];
- f44 = c[i + 3 + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f22 = C_ARRAY_ELEM (i + 1, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f23 = C_ARRAY_ELEM (i + 1, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
+ f24 = C_ARRAY_ELEM (i + 1, j + 3);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
+ f32 = C_ARRAY_ELEM (i + 2, j + 1);
+ f42 = C_ARRAY_ELEM (i + 3, j + 1);
+ f33 = C_ARRAY_ELEM (i + 2, j + 2);
+ f43 = C_ARRAY_ELEM (i + 3, j + 2);
+ f34 = C_ARRAY_ELEM (i + 2, j + 3);
+ f44 = C_ARRAY_ELEM (i + 3, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f22 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f23 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f24 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f32 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f42 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f33 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f43 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f34 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f44 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + 1 + (j + 1) * c_dim1] = f22;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + 1 + (j + 2) * c_dim1] = f23;
- c[i + (j + 3) * c_dim1] = f14;
- c[i + 1 + (j + 3) * c_dim1] = f24;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
- c[i + 2 + (j + 1) * c_dim1] = f32;
- c[i + 3 + (j + 1) * c_dim1] = f42;
- c[i + 2 + (j + 2) * c_dim1] = f33;
- c[i + 3 + (j + 2) * c_dim1] = f43;
- c[i + 2 + (j + 3) * c_dim1] = f34;
- c[i + 3 + (j + 3) * c_dim1] = f44;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i + 1, j + 1) = f22;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i + 1, j + 2) = f23;
+ C_ARRAY_ELEM (i, j + 3) = f14;
+ C_ARRAY_ELEM (i + 1, j + 3) = f24;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
+ C_ARRAY_ELEM (i + 2, j + 1) = f32;
+ C_ARRAY_ELEM (i + 3, j + 1) = f42;
+ C_ARRAY_ELEM (i + 2, j + 2) = f33;
+ C_ARRAY_ELEM (i + 3, j + 2) = f43;
+ C_ARRAY_ELEM (i + 2, j + 3) = f34;
+ C_ARRAY_ELEM (i + 3, j + 3) = f44;
}
if (uisec < isec)
{
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 1) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 2) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 3) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + (j + 3) * c_dim1] = f14;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i, j + 3) = f14;
}
}
}
@@ -1647,38 +1693,38 @@ matmul_c16_avx512f (gfc_array_c16 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
}
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
+ C_ARRAY_ELEM (i, j) = f11;
}
}
}
@@ -1687,6 +1733,9 @@ matmul_c16_avx512f (gfc_array_c16 * const restrict retarray,
}
free(t1);
return;
+#undef A_ARRAY_ELEM
+#undef B_ARRAY_ELEM
+#undef C_ARRAY_ELEM
}
else if (rxstride == 1 && aystride == 1 && bxstride == 1)
{
@@ -1699,11 +1748,11 @@ matmul_c16_avx512f (gfc_array_c16 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_16) 0;
for (n = 0; n < count; n++)
s += abase_x[n] * bbase_y[n];
@@ -1718,11 +1767,11 @@ matmul_c16_avx512f (gfc_array_c16 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_16) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n];
- dest[y*rystride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n) * bbase_y[n];
+ ARRAY_ELEM_AT_OFFSET (dest, y * rystride_bytes) = s;
}
}
}
@@ -1733,26 +1782,27 @@ matmul_c16_avx512f (gfc_array_c16 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_16) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n*bxstride];
- dest[y*rxstride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
}
}
else if (axstride < aystride)
{
for (y = 0; y < ycount; y++)
for (x = 0; x < xcount; x++)
- dest[x*rxstride + y*rystride] = (GFC_COMPLEX_16)0;
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y) = (GFC_COMPLEX_16)0;
for (y = 0; y < ycount; y++)
for (n = 0; n < count; n++)
for (x = 0; x < xcount; x++)
/* dest[x,y] += a[x,n] * b[n,y] */
- dest[x*rxstride + y*rystride] +=
- abase[x*axstride + n*aystride] *
- bbase[n*bxstride + y*bystride];
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y)
+ += GFC_DESCRIPTOR2_ELEM (a, x, n)
+ * GFC_DESCRIPTOR2_ELEM (b, n, y);
}
else
{
@@ -1763,15 +1813,16 @@ matmul_c16_avx512f (gfc_array_c16 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_16) 0;
for (n = 0; n < count; n++)
- s += abase_x[n*aystride] * bbase_y[n*bxstride];
- dest_y[x*rxstride] = s;
+ s += ARRAY_ELEM_AT_OFFSET (abase_x, n * aystride_bytes)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ ARRAY_ELEM_AT_OFFSET (dest_y, x * rxstride_bytes) = s;
}
}
}
@@ -1812,7 +1863,8 @@ matmul_c16_vanilla (gfc_array_c16 * const restrict retarray,
index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
index_type x, y, n, count, xcount, ycount;
- index_type aystride_bytes, bystride_bytes, rystride_bytes;
+ index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
+ rxstride_bytes, rystride_bytes;
assert (GFC_DESCRIPTOR_RANK (a) == 2
|| GFC_DESCRIPTOR_RANK (b) == 2);
@@ -1901,12 +1953,13 @@ matmul_c16_vanilla (gfc_array_c16 * const restrict retarray,
either as a row or a column matrix. We want both cases to
work. */
rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
+ rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
}
else
{
rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
+ rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
}
@@ -1915,6 +1968,7 @@ matmul_c16_vanilla (gfc_array_c16 * const restrict retarray,
/* Treat it as a a row matrix A[1,count]. */
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = 1;
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = sizeof (GFC_COMPLEX_16);
xcount = 1;
@@ -1924,6 +1978,7 @@ matmul_c16_vanilla (gfc_array_c16 * const restrict retarray,
{
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = GFC_DESCRIPTOR_STRIDE(a,1);
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,1);
count = GFC_DESCRIPTOR_EXTENT(a,1);
@@ -1942,17 +1997,20 @@ matmul_c16_vanilla (gfc_array_c16 * const restrict retarray,
{
/* Treat it as a column matrix B[count,1] */
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
/* bystride should never be used for 1-dimensional b.
The value is only used for calculation of the
memory by the buffer. */
bystride = 256;
+ bystride_bytes = 99999999;
ycount = 1;
}
else
{
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
bystride = GFC_DESCRIPTOR_STRIDE(b,1);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
bystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,1);
ycount = GFC_DESCRIPTOR_EXTENT(b,1);
}
@@ -2012,12 +2070,11 @@ matmul_c16_vanilla (gfc_array_c16 * const restrict retarray,
from netlib.org, translated to C, and modified for matmul.m4. */
- const GFC_COMPLEX_16 *a, *b;
GFC_COMPLEX_16 *c;
const index_type m = xcount, n = ycount, k = count;
/* System generated locals */
- index_type a_dim1, b_dim1, c_dim1,
+ index_type a_dim1, b_dim1,
i1, i2, i3, i4, i5, i6;
/* Local variables */
@@ -2027,19 +2084,25 @@ matmul_c16_vanilla (gfc_array_c16 * const restrict retarray,
index_type isec, jsec, lsec, uisec, ujsec, ulsec;
GFC_COMPLEX_16 *t1;
- a = abase;
- b = bbase;
c = retarray->base_addr;
/* Parameter adjustments */
- c_dim1 = rystride;
a_dim1 = aystride;
b_dim1 = bystride;
- /* Empty c first. */
+#define A_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (abase, (i) * sizeof (GFC_COMPLEX_16) + (j) * aystride_bytes))
+
+#define B_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (bbase, (i) * sizeof (GFC_COMPLEX_16) + (j) * bystride_bytes))
+
+#define C_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (c, (i) * sizeof (GFC_COMPLEX_16) + (j) * rystride_bytes))
+
+ /* Empty result first. */
for (j=0; j<n; j++)
for (i=0; i<m; i++)
- c[i + j * c_dim1] = (GFC_COMPLEX_16)0;
+ C_ARRAY_ELEM (i, j) = (GFC_COMPLEX_16)0;
/* Early exit if possible */
if (m == 0 || n == 0 || k == 0)
@@ -2091,20 +2154,20 @@ matmul_c16_vanilla (gfc_array_c16 * const restrict retarray,
for (i = ii; i < i5; i += 2)
{
t1[l - ll + 1 + ((i - ii + 1) << 8) - 257] =
- a[i + l * a_dim1];
+ A_ARRAY_ELEM (i, l);
t1[l - ll + 2 + ((i - ii + 1) << 8) - 257] =
- a[i + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i, l + 1);
t1[l - ll + 1 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + l * a_dim1];
+ A_ARRAY_ELEM (i + 1, l);
t1[l - ll + 2 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i + 1, l + 1);
}
if (uisec < isec)
{
t1[l - ll + 1 + (isec << 8) - 257] =
- a[ii + isec - 1 + l * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l);
t1[l - ll + 2 + (isec << 8) - 257] =
- a[ii + isec - 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l + 1);
}
}
if (ulsec < lsec)
@@ -2113,7 +2176,7 @@ matmul_c16_vanilla (gfc_array_c16 * const restrict retarray,
for (i = ii; i< i4; ++i)
{
t1[lsec + ((i - ii + 1) << 8) - 257] =
- a[i + (ll + lsec - 1) * a_dim1];
+ A_ARRAY_ELEM (i, ll + lsec - 1);
}
}
@@ -2124,100 +2187,100 @@ matmul_c16_vanilla (gfc_array_c16 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f22 = c[i + 1 + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f23 = c[i + 1 + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
- f24 = c[i + 1 + (j + 3) * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
- f32 = c[i + 2 + (j + 1) * c_dim1];
- f42 = c[i + 3 + (j + 1) * c_dim1];
- f33 = c[i + 2 + (j + 2) * c_dim1];
- f43 = c[i + 3 + (j + 2) * c_dim1];
- f34 = c[i + 2 + (j + 3) * c_dim1];
- f44 = c[i + 3 + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f22 = C_ARRAY_ELEM (i + 1, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f23 = C_ARRAY_ELEM (i + 1, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
+ f24 = C_ARRAY_ELEM (i + 1, j + 3);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
+ f32 = C_ARRAY_ELEM (i + 2, j + 1);
+ f42 = C_ARRAY_ELEM (i + 3, j + 1);
+ f33 = C_ARRAY_ELEM (i + 2, j + 2);
+ f43 = C_ARRAY_ELEM (i + 3, j + 2);
+ f34 = C_ARRAY_ELEM (i + 2, j + 3);
+ f44 = C_ARRAY_ELEM (i + 3, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f22 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f23 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f24 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f32 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f42 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f33 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f43 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f34 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f44 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + 1 + (j + 1) * c_dim1] = f22;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + 1 + (j + 2) * c_dim1] = f23;
- c[i + (j + 3) * c_dim1] = f14;
- c[i + 1 + (j + 3) * c_dim1] = f24;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
- c[i + 2 + (j + 1) * c_dim1] = f32;
- c[i + 3 + (j + 1) * c_dim1] = f42;
- c[i + 2 + (j + 2) * c_dim1] = f33;
- c[i + 3 + (j + 2) * c_dim1] = f43;
- c[i + 2 + (j + 3) * c_dim1] = f34;
- c[i + 3 + (j + 3) * c_dim1] = f44;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i + 1, j + 1) = f22;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i + 1, j + 2) = f23;
+ C_ARRAY_ELEM (i, j + 3) = f14;
+ C_ARRAY_ELEM (i + 1, j + 3) = f24;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
+ C_ARRAY_ELEM (i + 2, j + 1) = f32;
+ C_ARRAY_ELEM (i + 3, j + 1) = f42;
+ C_ARRAY_ELEM (i + 2, j + 2) = f33;
+ C_ARRAY_ELEM (i + 3, j + 2) = f43;
+ C_ARRAY_ELEM (i + 2, j + 3) = f34;
+ C_ARRAY_ELEM (i + 3, j + 3) = f44;
}
if (uisec < isec)
{
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 1) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 2) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 3) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + (j + 3) * c_dim1] = f14;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i, j + 3) = f14;
}
}
}
@@ -2229,38 +2292,38 @@ matmul_c16_vanilla (gfc_array_c16 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
}
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
+ C_ARRAY_ELEM (i, j) = f11;
}
}
}
@@ -2269,6 +2332,9 @@ matmul_c16_vanilla (gfc_array_c16 * const restrict retarray,
}
free(t1);
return;
+#undef A_ARRAY_ELEM
+#undef B_ARRAY_ELEM
+#undef C_ARRAY_ELEM
}
else if (rxstride == 1 && aystride == 1 && bxstride == 1)
{
@@ -2281,11 +2347,11 @@ matmul_c16_vanilla (gfc_array_c16 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_16) 0;
for (n = 0; n < count; n++)
s += abase_x[n] * bbase_y[n];
@@ -2300,11 +2366,11 @@ matmul_c16_vanilla (gfc_array_c16 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_16) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n];
- dest[y*rystride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n) * bbase_y[n];
+ ARRAY_ELEM_AT_OFFSET (dest, y * rystride_bytes) = s;
}
}
}
@@ -2315,26 +2381,27 @@ matmul_c16_vanilla (gfc_array_c16 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_16) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n*bxstride];
- dest[y*rxstride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
}
}
else if (axstride < aystride)
{
for (y = 0; y < ycount; y++)
for (x = 0; x < xcount; x++)
- dest[x*rxstride + y*rystride] = (GFC_COMPLEX_16)0;
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y) = (GFC_COMPLEX_16)0;
for (y = 0; y < ycount; y++)
for (n = 0; n < count; n++)
for (x = 0; x < xcount; x++)
/* dest[x,y] += a[x,n] * b[n,y] */
- dest[x*rxstride + y*rystride] +=
- abase[x*axstride + n*aystride] *
- bbase[n*bxstride + y*bystride];
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y)
+ += GFC_DESCRIPTOR2_ELEM (a, x, n)
+ * GFC_DESCRIPTOR2_ELEM (b, n, y);
}
else
{
@@ -2345,15 +2412,16 @@ matmul_c16_vanilla (gfc_array_c16 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_16) 0;
for (n = 0; n < count; n++)
- s += abase_x[n*aystride] * bbase_y[n*bxstride];
- dest_y[x*rxstride] = s;
+ s += ARRAY_ELEM_AT_OFFSET (abase_x, n * aystride_bytes)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ ARRAY_ELEM_AT_OFFSET (dest_y, x * rxstride_bytes) = s;
}
}
}
@@ -2453,7 +2521,8 @@ matmul_c16 (gfc_array_c16 * const restrict retarray,
index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
index_type x, y, n, count, xcount, ycount;
- index_type aystride_bytes, bystride_bytes, rystride_bytes;
+ index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
+ rxstride_bytes, rystride_bytes;
assert (GFC_DESCRIPTOR_RANK (a) == 2
|| GFC_DESCRIPTOR_RANK (b) == 2);
@@ -2542,12 +2611,13 @@ matmul_c16 (gfc_array_c16 * const restrict retarray,
either as a row or a column matrix. We want both cases to
work. */
rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
+ rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
}
else
{
rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
+ rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
}
@@ -2556,6 +2626,7 @@ matmul_c16 (gfc_array_c16 * const restrict retarray,
/* Treat it as a a row matrix A[1,count]. */
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = 1;
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = sizeof (GFC_COMPLEX_16);
xcount = 1;
@@ -2565,6 +2636,7 @@ matmul_c16 (gfc_array_c16 * const restrict retarray,
{
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = GFC_DESCRIPTOR_STRIDE(a,1);
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,1);
count = GFC_DESCRIPTOR_EXTENT(a,1);
@@ -2583,17 +2655,20 @@ matmul_c16 (gfc_array_c16 * const restrict retarray,
{
/* Treat it as a column matrix B[count,1] */
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
/* bystride should never be used for 1-dimensional b.
The value is only used for calculation of the
memory by the buffer. */
bystride = 256;
+ bystride_bytes = 99999999;
ycount = 1;
}
else
{
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
bystride = GFC_DESCRIPTOR_STRIDE(b,1);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
bystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,1);
ycount = GFC_DESCRIPTOR_EXTENT(b,1);
}
@@ -2653,12 +2728,11 @@ matmul_c16 (gfc_array_c16 * const restrict retarray,
from netlib.org, translated to C, and modified for matmul.m4. */
- const GFC_COMPLEX_16 *a, *b;
GFC_COMPLEX_16 *c;
const index_type m = xcount, n = ycount, k = count;
/* System generated locals */
- index_type a_dim1, b_dim1, c_dim1,
+ index_type a_dim1, b_dim1,
i1, i2, i3, i4, i5, i6;
/* Local variables */
@@ -2668,19 +2742,25 @@ matmul_c16 (gfc_array_c16 * const restrict retarray,
index_type isec, jsec, lsec, uisec, ujsec, ulsec;
GFC_COMPLEX_16 *t1;
- a = abase;
- b = bbase;
c = retarray->base_addr;
/* Parameter adjustments */
- c_dim1 = rystride;
a_dim1 = aystride;
b_dim1 = bystride;
- /* Empty c first. */
+#define A_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (abase, (i) * sizeof (GFC_COMPLEX_16) + (j) * aystride_bytes))
+
+#define B_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (bbase, (i) * sizeof (GFC_COMPLEX_16) + (j) * bystride_bytes))
+
+#define C_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (c, (i) * sizeof (GFC_COMPLEX_16) + (j) * rystride_bytes))
+
+ /* Empty result first. */
for (j=0; j<n; j++)
for (i=0; i<m; i++)
- c[i + j * c_dim1] = (GFC_COMPLEX_16)0;
+ C_ARRAY_ELEM (i, j) = (GFC_COMPLEX_16)0;
/* Early exit if possible */
if (m == 0 || n == 0 || k == 0)
@@ -2732,20 +2812,20 @@ matmul_c16 (gfc_array_c16 * const restrict retarray,
for (i = ii; i < i5; i += 2)
{
t1[l - ll + 1 + ((i - ii + 1) << 8) - 257] =
- a[i + l * a_dim1];
+ A_ARRAY_ELEM (i, l);
t1[l - ll + 2 + ((i - ii + 1) << 8) - 257] =
- a[i + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i, l + 1);
t1[l - ll + 1 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + l * a_dim1];
+ A_ARRAY_ELEM (i + 1, l);
t1[l - ll + 2 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i + 1, l + 1);
}
if (uisec < isec)
{
t1[l - ll + 1 + (isec << 8) - 257] =
- a[ii + isec - 1 + l * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l);
t1[l - ll + 2 + (isec << 8) - 257] =
- a[ii + isec - 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l + 1);
}
}
if (ulsec < lsec)
@@ -2754,7 +2834,7 @@ matmul_c16 (gfc_array_c16 * const restrict retarray,
for (i = ii; i< i4; ++i)
{
t1[lsec + ((i - ii + 1) << 8) - 257] =
- a[i + (ll + lsec - 1) * a_dim1];
+ A_ARRAY_ELEM (i, ll + lsec - 1);
}
}
@@ -2765,100 +2845,100 @@ matmul_c16 (gfc_array_c16 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f22 = c[i + 1 + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f23 = c[i + 1 + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
- f24 = c[i + 1 + (j + 3) * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
- f32 = c[i + 2 + (j + 1) * c_dim1];
- f42 = c[i + 3 + (j + 1) * c_dim1];
- f33 = c[i + 2 + (j + 2) * c_dim1];
- f43 = c[i + 3 + (j + 2) * c_dim1];
- f34 = c[i + 2 + (j + 3) * c_dim1];
- f44 = c[i + 3 + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f22 = C_ARRAY_ELEM (i + 1, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f23 = C_ARRAY_ELEM (i + 1, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
+ f24 = C_ARRAY_ELEM (i + 1, j + 3);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
+ f32 = C_ARRAY_ELEM (i + 2, j + 1);
+ f42 = C_ARRAY_ELEM (i + 3, j + 1);
+ f33 = C_ARRAY_ELEM (i + 2, j + 2);
+ f43 = C_ARRAY_ELEM (i + 3, j + 2);
+ f34 = C_ARRAY_ELEM (i + 2, j + 3);
+ f44 = C_ARRAY_ELEM (i + 3, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f22 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f23 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f24 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f32 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f42 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f33 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f43 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f34 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f44 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + 1 + (j + 1) * c_dim1] = f22;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + 1 + (j + 2) * c_dim1] = f23;
- c[i + (j + 3) * c_dim1] = f14;
- c[i + 1 + (j + 3) * c_dim1] = f24;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
- c[i + 2 + (j + 1) * c_dim1] = f32;
- c[i + 3 + (j + 1) * c_dim1] = f42;
- c[i + 2 + (j + 2) * c_dim1] = f33;
- c[i + 3 + (j + 2) * c_dim1] = f43;
- c[i + 2 + (j + 3) * c_dim1] = f34;
- c[i + 3 + (j + 3) * c_dim1] = f44;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i + 1, j + 1) = f22;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i + 1, j + 2) = f23;
+ C_ARRAY_ELEM (i, j + 3) = f14;
+ C_ARRAY_ELEM (i + 1, j + 3) = f24;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
+ C_ARRAY_ELEM (i + 2, j + 1) = f32;
+ C_ARRAY_ELEM (i + 3, j + 1) = f42;
+ C_ARRAY_ELEM (i + 2, j + 2) = f33;
+ C_ARRAY_ELEM (i + 3, j + 2) = f43;
+ C_ARRAY_ELEM (i + 2, j + 3) = f34;
+ C_ARRAY_ELEM (i + 3, j + 3) = f44;
}
if (uisec < isec)
{
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 1) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 2) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 3) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + (j + 3) * c_dim1] = f14;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i, j + 3) = f14;
}
}
}
@@ -2870,38 +2950,38 @@ matmul_c16 (gfc_array_c16 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
}
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
+ C_ARRAY_ELEM (i, j) = f11;
}
}
}
@@ -2910,6 +2990,9 @@ matmul_c16 (gfc_array_c16 * const restrict retarray,
}
free(t1);
return;
+#undef A_ARRAY_ELEM
+#undef B_ARRAY_ELEM
+#undef C_ARRAY_ELEM
}
else if (rxstride == 1 && aystride == 1 && bxstride == 1)
{
@@ -2922,11 +3005,11 @@ matmul_c16 (gfc_array_c16 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_16) 0;
for (n = 0; n < count; n++)
s += abase_x[n] * bbase_y[n];
@@ -2941,11 +3024,11 @@ matmul_c16 (gfc_array_c16 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_16) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n];
- dest[y*rystride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n) * bbase_y[n];
+ ARRAY_ELEM_AT_OFFSET (dest, y * rystride_bytes) = s;
}
}
}
@@ -2956,26 +3039,27 @@ matmul_c16 (gfc_array_c16 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_16) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n*bxstride];
- dest[y*rxstride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
}
}
else if (axstride < aystride)
{
for (y = 0; y < ycount; y++)
for (x = 0; x < xcount; x++)
- dest[x*rxstride + y*rystride] = (GFC_COMPLEX_16)0;
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y) = (GFC_COMPLEX_16)0;
for (y = 0; y < ycount; y++)
for (n = 0; n < count; n++)
for (x = 0; x < xcount; x++)
/* dest[x,y] += a[x,n] * b[n,y] */
- dest[x*rxstride + y*rystride] +=
- abase[x*axstride + n*aystride] *
- bbase[n*bxstride + y*bystride];
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y)
+ += GFC_DESCRIPTOR2_ELEM (a, x, n)
+ * GFC_DESCRIPTOR2_ELEM (b, n, y);
}
else
{
@@ -2986,15 +3070,16 @@ matmul_c16 (gfc_array_c16 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_16) 0;
for (n = 0; n < count; n++)
- s += abase_x[n*aystride] * bbase_y[n*bxstride];
- dest_y[x*rxstride] = s;
+ s += ARRAY_ELEM_AT_OFFSET (abase_x, n * aystride_bytes)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ ARRAY_ELEM_AT_OFFSET (dest_y, x * rxstride_bytes) = s;
}
}
}
diff --git a/libgfortran/generated/matmul_c17.c b/libgfortran/generated/matmul_c17.c
index e0f379005944..47b76385fb67 100644
--- a/libgfortran/generated/matmul_c17.c
+++ b/libgfortran/generated/matmul_c17.c
@@ -94,7 +94,8 @@ matmul_c17_avx (gfc_array_c17 * const restrict retarray,
index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
index_type x, y, n, count, xcount, ycount;
- index_type aystride_bytes, bystride_bytes, rystride_bytes;
+ index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
+ rxstride_bytes, rystride_bytes;
assert (GFC_DESCRIPTOR_RANK (a) == 2
|| GFC_DESCRIPTOR_RANK (b) == 2);
@@ -183,12 +184,13 @@ matmul_c17_avx (gfc_array_c17 * const restrict retarray,
either as a row or a column matrix. We want both cases to
work. */
rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
+ rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
}
else
{
rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
+ rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
}
@@ -197,6 +199,7 @@ matmul_c17_avx (gfc_array_c17 * const restrict retarray,
/* Treat it as a a row matrix A[1,count]. */
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = 1;
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = sizeof (GFC_COMPLEX_17);
xcount = 1;
@@ -206,6 +209,7 @@ matmul_c17_avx (gfc_array_c17 * const restrict retarray,
{
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = GFC_DESCRIPTOR_STRIDE(a,1);
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,1);
count = GFC_DESCRIPTOR_EXTENT(a,1);
@@ -224,17 +228,20 @@ matmul_c17_avx (gfc_array_c17 * const restrict retarray,
{
/* Treat it as a column matrix B[count,1] */
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
/* bystride should never be used for 1-dimensional b.
The value is only used for calculation of the
memory by the buffer. */
bystride = 256;
+ bystride_bytes = 99999999;
ycount = 1;
}
else
{
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
bystride = GFC_DESCRIPTOR_STRIDE(b,1);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
bystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,1);
ycount = GFC_DESCRIPTOR_EXTENT(b,1);
}
@@ -294,12 +301,11 @@ matmul_c17_avx (gfc_array_c17 * const restrict retarray,
from netlib.org, translated to C, and modified for matmul.m4. */
- const GFC_COMPLEX_17 *a, *b;
GFC_COMPLEX_17 *c;
const index_type m = xcount, n = ycount, k = count;
/* System generated locals */
- index_type a_dim1, b_dim1, c_dim1,
+ index_type a_dim1, b_dim1,
i1, i2, i3, i4, i5, i6;
/* Local variables */
@@ -309,19 +315,25 @@ matmul_c17_avx (gfc_array_c17 * const restrict retarray,
index_type isec, jsec, lsec, uisec, ujsec, ulsec;
GFC_COMPLEX_17 *t1;
- a = abase;
- b = bbase;
c = retarray->base_addr;
/* Parameter adjustments */
- c_dim1 = rystride;
a_dim1 = aystride;
b_dim1 = bystride;
- /* Empty c first. */
+#define A_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (abase, (i) * sizeof (GFC_COMPLEX_17) + (j) * aystride_bytes))
+
+#define B_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (bbase, (i) * sizeof (GFC_COMPLEX_17) + (j) * bystride_bytes))
+
+#define C_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (c, (i) * sizeof (GFC_COMPLEX_17) + (j) * rystride_bytes))
+
+ /* Empty result first. */
for (j=0; j<n; j++)
for (i=0; i<m; i++)
- c[i + j * c_dim1] = (GFC_COMPLEX_17)0;
+ C_ARRAY_ELEM (i, j) = (GFC_COMPLEX_17)0;
/* Early exit if possible */
if (m == 0 || n == 0 || k == 0)
@@ -373,20 +385,20 @@ matmul_c17_avx (gfc_array_c17 * const restrict retarray,
for (i = ii; i < i5; i += 2)
{
t1[l - ll + 1 + ((i - ii + 1) << 8) - 257] =
- a[i + l * a_dim1];
+ A_ARRAY_ELEM (i, l);
t1[l - ll + 2 + ((i - ii + 1) << 8) - 257] =
- a[i + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i, l + 1);
t1[l - ll + 1 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + l * a_dim1];
+ A_ARRAY_ELEM (i + 1, l);
t1[l - ll + 2 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i + 1, l + 1);
}
if (uisec < isec)
{
t1[l - ll + 1 + (isec << 8) - 257] =
- a[ii + isec - 1 + l * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l);
t1[l - ll + 2 + (isec << 8) - 257] =
- a[ii + isec - 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l + 1);
}
}
if (ulsec < lsec)
@@ -395,7 +407,7 @@ matmul_c17_avx (gfc_array_c17 * const restrict retarray,
for (i = ii; i< i4; ++i)
{
t1[lsec + ((i - ii + 1) << 8) - 257] =
- a[i + (ll + lsec - 1) * a_dim1];
+ A_ARRAY_ELEM (i, ll + lsec - 1);
}
}
@@ -406,100 +418,100 @@ matmul_c17_avx (gfc_array_c17 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f22 = c[i + 1 + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f23 = c[i + 1 + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
- f24 = c[i + 1 + (j + 3) * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
- f32 = c[i + 2 + (j + 1) * c_dim1];
- f42 = c[i + 3 + (j + 1) * c_dim1];
- f33 = c[i + 2 + (j + 2) * c_dim1];
- f43 = c[i + 3 + (j + 2) * c_dim1];
- f34 = c[i + 2 + (j + 3) * c_dim1];
- f44 = c[i + 3 + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f22 = C_ARRAY_ELEM (i + 1, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f23 = C_ARRAY_ELEM (i + 1, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
+ f24 = C_ARRAY_ELEM (i + 1, j + 3);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
+ f32 = C_ARRAY_ELEM (i + 2, j + 1);
+ f42 = C_ARRAY_ELEM (i + 3, j + 1);
+ f33 = C_ARRAY_ELEM (i + 2, j + 2);
+ f43 = C_ARRAY_ELEM (i + 3, j + 2);
+ f34 = C_ARRAY_ELEM (i + 2, j + 3);
+ f44 = C_ARRAY_ELEM (i + 3, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f22 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f23 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f24 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f32 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f42 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f33 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f43 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f34 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f44 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + 1 + (j + 1) * c_dim1] = f22;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + 1 + (j + 2) * c_dim1] = f23;
- c[i + (j + 3) * c_dim1] = f14;
- c[i + 1 + (j + 3) * c_dim1] = f24;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
- c[i + 2 + (j + 1) * c_dim1] = f32;
- c[i + 3 + (j + 1) * c_dim1] = f42;
- c[i + 2 + (j + 2) * c_dim1] = f33;
- c[i + 3 + (j + 2) * c_dim1] = f43;
- c[i + 2 + (j + 3) * c_dim1] = f34;
- c[i + 3 + (j + 3) * c_dim1] = f44;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i + 1, j + 1) = f22;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i + 1, j + 2) = f23;
+ C_ARRAY_ELEM (i, j + 3) = f14;
+ C_ARRAY_ELEM (i + 1, j + 3) = f24;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
+ C_ARRAY_ELEM (i + 2, j + 1) = f32;
+ C_ARRAY_ELEM (i + 3, j + 1) = f42;
+ C_ARRAY_ELEM (i + 2, j + 2) = f33;
+ C_ARRAY_ELEM (i + 3, j + 2) = f43;
+ C_ARRAY_ELEM (i + 2, j + 3) = f34;
+ C_ARRAY_ELEM (i + 3, j + 3) = f44;
}
if (uisec < isec)
{
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 1) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 2) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 3) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + (j + 3) * c_dim1] = f14;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i, j + 3) = f14;
}
}
}
@@ -511,38 +523,38 @@ matmul_c17_avx (gfc_array_c17 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
}
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
+ C_ARRAY_ELEM (i, j) = f11;
}
}
}
@@ -551,6 +563,9 @@ matmul_c17_avx (gfc_array_c17 * const restrict retarray,
}
free(t1);
return;
+#undef A_ARRAY_ELEM
+#undef B_ARRAY_ELEM
+#undef C_ARRAY_ELEM
}
else if (rxstride == 1 && aystride == 1 && bxstride == 1)
{
@@ -563,11 +578,11 @@ matmul_c17_avx (gfc_array_c17 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_17) 0;
for (n = 0; n < count; n++)
s += abase_x[n] * bbase_y[n];
@@ -582,11 +597,11 @@ matmul_c17_avx (gfc_array_c17 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_17) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n];
- dest[y*rystride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n) * bbase_y[n];
+ ARRAY_ELEM_AT_OFFSET (dest, y * rystride_bytes) = s;
}
}
}
@@ -597,26 +612,27 @@ matmul_c17_avx (gfc_array_c17 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_17) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n*bxstride];
- dest[y*rxstride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
}
}
else if (axstride < aystride)
{
for (y = 0; y < ycount; y++)
for (x = 0; x < xcount; x++)
- dest[x*rxstride + y*rystride] = (GFC_COMPLEX_17)0;
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y) = (GFC_COMPLEX_17)0;
for (y = 0; y < ycount; y++)
for (n = 0; n < count; n++)
for (x = 0; x < xcount; x++)
/* dest[x,y] += a[x,n] * b[n,y] */
- dest[x*rxstride + y*rystride] +=
- abase[x*axstride + n*aystride] *
- bbase[n*bxstride + y*bystride];
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y)
+ += GFC_DESCRIPTOR2_ELEM (a, x, n)
+ * GFC_DESCRIPTOR2_ELEM (b, n, y);
}
else
{
@@ -627,15 +643,16 @@ matmul_c17_avx (gfc_array_c17 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_17) 0;
for (n = 0; n < count; n++)
- s += abase_x[n*aystride] * bbase_y[n*bxstride];
- dest_y[x*rxstride] = s;
+ s += ARRAY_ELEM_AT_OFFSET (abase_x, n * aystride_bytes)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ ARRAY_ELEM_AT_OFFSET (dest_y, x * rxstride_bytes) = s;
}
}
}
@@ -662,7 +679,8 @@ matmul_c17_avx2 (gfc_array_c17 * const restrict retarray,
index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
index_type x, y, n, count, xcount, ycount;
- index_type aystride_bytes, bystride_bytes, rystride_bytes;
+ index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
+ rxstride_bytes, rystride_bytes;
assert (GFC_DESCRIPTOR_RANK (a) == 2
|| GFC_DESCRIPTOR_RANK (b) == 2);
@@ -751,12 +769,13 @@ matmul_c17_avx2 (gfc_array_c17 * const restrict retarray,
either as a row or a column matrix. We want both cases to
work. */
rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
+ rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
}
else
{
rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
+ rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
}
@@ -765,6 +784,7 @@ matmul_c17_avx2 (gfc_array_c17 * const restrict retarray,
/* Treat it as a a row matrix A[1,count]. */
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = 1;
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = sizeof (GFC_COMPLEX_17);
xcount = 1;
@@ -774,6 +794,7 @@ matmul_c17_avx2 (gfc_array_c17 * const restrict retarray,
{
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = GFC_DESCRIPTOR_STRIDE(a,1);
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,1);
count = GFC_DESCRIPTOR_EXTENT(a,1);
@@ -792,17 +813,20 @@ matmul_c17_avx2 (gfc_array_c17 * const restrict retarray,
{
/* Treat it as a column matrix B[count,1] */
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
/* bystride should never be used for 1-dimensional b.
The value is only used for calculation of the
memory by the buffer. */
bystride = 256;
+ bystride_bytes = 99999999;
ycount = 1;
}
else
{
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
bystride = GFC_DESCRIPTOR_STRIDE(b,1);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
bystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,1);
ycount = GFC_DESCRIPTOR_EXTENT(b,1);
}
@@ -862,12 +886,11 @@ matmul_c17_avx2 (gfc_array_c17 * const restrict retarray,
from netlib.org, translated to C, and modified for matmul.m4. */
- const GFC_COMPLEX_17 *a, *b;
GFC_COMPLEX_17 *c;
const index_type m = xcount, n = ycount, k = count;
/* System generated locals */
- index_type a_dim1, b_dim1, c_dim1,
+ index_type a_dim1, b_dim1,
i1, i2, i3, i4, i5, i6;
/* Local variables */
@@ -877,19 +900,25 @@ matmul_c17_avx2 (gfc_array_c17 * const restrict retarray,
index_type isec, jsec, lsec, uisec, ujsec, ulsec;
GFC_COMPLEX_17 *t1;
- a = abase;
- b = bbase;
c = retarray->base_addr;
/* Parameter adjustments */
- c_dim1 = rystride;
a_dim1 = aystride;
b_dim1 = bystride;
- /* Empty c first. */
+#define A_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (abase, (i) * sizeof (GFC_COMPLEX_17) + (j) * aystride_bytes))
+
+#define B_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (bbase, (i) * sizeof (GFC_COMPLEX_17) + (j) * bystride_bytes))
+
+#define C_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (c, (i) * sizeof (GFC_COMPLEX_17) + (j) * rystride_bytes))
+
+ /* Empty result first. */
for (j=0; j<n; j++)
for (i=0; i<m; i++)
- c[i + j * c_dim1] = (GFC_COMPLEX_17)0;
+ C_ARRAY_ELEM (i, j) = (GFC_COMPLEX_17)0;
/* Early exit if possible */
if (m == 0 || n == 0 || k == 0)
@@ -941,20 +970,20 @@ matmul_c17_avx2 (gfc_array_c17 * const restrict retarray,
for (i = ii; i < i5; i += 2)
{
t1[l - ll + 1 + ((i - ii + 1) << 8) - 257] =
- a[i + l * a_dim1];
+ A_ARRAY_ELEM (i, l);
t1[l - ll + 2 + ((i - ii + 1) << 8) - 257] =
- a[i + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i, l + 1);
t1[l - ll + 1 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + l * a_dim1];
+ A_ARRAY_ELEM (i + 1, l);
t1[l - ll + 2 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i + 1, l + 1);
}
if (uisec < isec)
{
t1[l - ll + 1 + (isec << 8) - 257] =
- a[ii + isec - 1 + l * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l);
t1[l - ll + 2 + (isec << 8) - 257] =
- a[ii + isec - 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l + 1);
}
}
if (ulsec < lsec)
@@ -963,7 +992,7 @@ matmul_c17_avx2 (gfc_array_c17 * const restrict retarray,
for (i = ii; i< i4; ++i)
{
t1[lsec + ((i - ii + 1) << 8) - 257] =
- a[i + (ll + lsec - 1) * a_dim1];
+ A_ARRAY_ELEM (i, ll + lsec - 1);
}
}
@@ -974,100 +1003,100 @@ matmul_c17_avx2 (gfc_array_c17 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f22 = c[i + 1 + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f23 = c[i + 1 + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
- f24 = c[i + 1 + (j + 3) * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
- f32 = c[i + 2 + (j + 1) * c_dim1];
- f42 = c[i + 3 + (j + 1) * c_dim1];
- f33 = c[i + 2 + (j + 2) * c_dim1];
- f43 = c[i + 3 + (j + 2) * c_dim1];
- f34 = c[i + 2 + (j + 3) * c_dim1];
- f44 = c[i + 3 + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f22 = C_ARRAY_ELEM (i + 1, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f23 = C_ARRAY_ELEM (i + 1, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
+ f24 = C_ARRAY_ELEM (i + 1, j + 3);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
+ f32 = C_ARRAY_ELEM (i + 2, j + 1);
+ f42 = C_ARRAY_ELEM (i + 3, j + 1);
+ f33 = C_ARRAY_ELEM (i + 2, j + 2);
+ f43 = C_ARRAY_ELEM (i + 3, j + 2);
+ f34 = C_ARRAY_ELEM (i + 2, j + 3);
+ f44 = C_ARRAY_ELEM (i + 3, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f22 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f23 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f24 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f32 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f42 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f33 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f43 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f34 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f44 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + 1 + (j + 1) * c_dim1] = f22;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + 1 + (j + 2) * c_dim1] = f23;
- c[i + (j + 3) * c_dim1] = f14;
- c[i + 1 + (j + 3) * c_dim1] = f24;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
- c[i + 2 + (j + 1) * c_dim1] = f32;
- c[i + 3 + (j + 1) * c_dim1] = f42;
- c[i + 2 + (j + 2) * c_dim1] = f33;
- c[i + 3 + (j + 2) * c_dim1] = f43;
- c[i + 2 + (j + 3) * c_dim1] = f34;
- c[i + 3 + (j + 3) * c_dim1] = f44;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i + 1, j + 1) = f22;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i + 1, j + 2) = f23;
+ C_ARRAY_ELEM (i, j + 3) = f14;
+ C_ARRAY_ELEM (i + 1, j + 3) = f24;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
+ C_ARRAY_ELEM (i + 2, j + 1) = f32;
+ C_ARRAY_ELEM (i + 3, j + 1) = f42;
+ C_ARRAY_ELEM (i + 2, j + 2) = f33;
+ C_ARRAY_ELEM (i + 3, j + 2) = f43;
+ C_ARRAY_ELEM (i + 2, j + 3) = f34;
+ C_ARRAY_ELEM (i + 3, j + 3) = f44;
}
if (uisec < isec)
{
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 1) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 2) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 3) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + (j + 3) * c_dim1] = f14;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i, j + 3) = f14;
}
}
}
@@ -1079,38 +1108,38 @@ matmul_c17_avx2 (gfc_array_c17 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
}
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
+ C_ARRAY_ELEM (i, j) = f11;
}
}
}
@@ -1119,6 +1148,9 @@ matmul_c17_avx2 (gfc_array_c17 * const restrict retarray,
}
free(t1);
return;
+#undef A_ARRAY_ELEM
+#undef B_ARRAY_ELEM
+#undef C_ARRAY_ELEM
}
else if (rxstride == 1 && aystride == 1 && bxstride == 1)
{
@@ -1131,11 +1163,11 @@ matmul_c17_avx2 (gfc_array_c17 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_17) 0;
for (n = 0; n < count; n++)
s += abase_x[n] * bbase_y[n];
@@ -1150,11 +1182,11 @@ matmul_c17_avx2 (gfc_array_c17 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_17) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n];
- dest[y*rystride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n) * bbase_y[n];
+ ARRAY_ELEM_AT_OFFSET (dest, y * rystride_bytes) = s;
}
}
}
@@ -1165,26 +1197,27 @@ matmul_c17_avx2 (gfc_array_c17 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_17) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n*bxstride];
- dest[y*rxstride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
}
}
else if (axstride < aystride)
{
for (y = 0; y < ycount; y++)
for (x = 0; x < xcount; x++)
- dest[x*rxstride + y*rystride] = (GFC_COMPLEX_17)0;
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y) = (GFC_COMPLEX_17)0;
for (y = 0; y < ycount; y++)
for (n = 0; n < count; n++)
for (x = 0; x < xcount; x++)
/* dest[x,y] += a[x,n] * b[n,y] */
- dest[x*rxstride + y*rystride] +=
- abase[x*axstride + n*aystride] *
- bbase[n*bxstride + y*bystride];
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y)
+ += GFC_DESCRIPTOR2_ELEM (a, x, n)
+ * GFC_DESCRIPTOR2_ELEM (b, n, y);
}
else
{
@@ -1195,15 +1228,16 @@ matmul_c17_avx2 (gfc_array_c17 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_17) 0;
for (n = 0; n < count; n++)
- s += abase_x[n*aystride] * bbase_y[n*bxstride];
- dest_y[x*rxstride] = s;
+ s += ARRAY_ELEM_AT_OFFSET (abase_x, n * aystride_bytes)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ ARRAY_ELEM_AT_OFFSET (dest_y, x * rxstride_bytes) = s;
}
}
}
@@ -1230,7 +1264,8 @@ matmul_c17_avx512f (gfc_array_c17 * const restrict retarray,
index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
index_type x, y, n, count, xcount, ycount;
- index_type aystride_bytes, bystride_bytes, rystride_bytes;
+ index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
+ rxstride_bytes, rystride_bytes;
assert (GFC_DESCRIPTOR_RANK (a) == 2
|| GFC_DESCRIPTOR_RANK (b) == 2);
@@ -1319,12 +1354,13 @@ matmul_c17_avx512f (gfc_array_c17 * const restrict retarray,
either as a row or a column matrix. We want both cases to
work. */
rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
+ rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
}
else
{
rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
+ rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
}
@@ -1333,6 +1369,7 @@ matmul_c17_avx512f (gfc_array_c17 * const restrict retarray,
/* Treat it as a a row matrix A[1,count]. */
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = 1;
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = sizeof (GFC_COMPLEX_17);
xcount = 1;
@@ -1342,6 +1379,7 @@ matmul_c17_avx512f (gfc_array_c17 * const restrict retarray,
{
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = GFC_DESCRIPTOR_STRIDE(a,1);
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,1);
count = GFC_DESCRIPTOR_EXTENT(a,1);
@@ -1360,17 +1398,20 @@ matmul_c17_avx512f (gfc_array_c17 * const restrict retarray,
{
/* Treat it as a column matrix B[count,1] */
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
/* bystride should never be used for 1-dimensional b.
The value is only used for calculation of the
memory by the buffer. */
bystride = 256;
+ bystride_bytes = 99999999;
ycount = 1;
}
else
{
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
bystride = GFC_DESCRIPTOR_STRIDE(b,1);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
bystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,1);
ycount = GFC_DESCRIPTOR_EXTENT(b,1);
}
@@ -1430,12 +1471,11 @@ matmul_c17_avx512f (gfc_array_c17 * const restrict retarray,
from netlib.org, translated to C, and modified for matmul.m4. */
- const GFC_COMPLEX_17 *a, *b;
GFC_COMPLEX_17 *c;
const index_type m = xcount, n = ycount, k = count;
/* System generated locals */
- index_type a_dim1, b_dim1, c_dim1,
+ index_type a_dim1, b_dim1,
i1, i2, i3, i4, i5, i6;
/* Local variables */
@@ -1445,19 +1485,25 @@ matmul_c17_avx512f (gfc_array_c17 * const restrict retarray,
index_type isec, jsec, lsec, uisec, ujsec, ulsec;
GFC_COMPLEX_17 *t1;
- a = abase;
- b = bbase;
c = retarray->base_addr;
/* Parameter adjustments */
- c_dim1 = rystride;
a_dim1 = aystride;
b_dim1 = bystride;
- /* Empty c first. */
+#define A_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (abase, (i) * sizeof (GFC_COMPLEX_17) + (j) * aystride_bytes))
+
+#define B_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (bbase, (i) * sizeof (GFC_COMPLEX_17) + (j) * bystride_bytes))
+
+#define C_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (c, (i) * sizeof (GFC_COMPLEX_17) + (j) * rystride_bytes))
+
+ /* Empty result first. */
for (j=0; j<n; j++)
for (i=0; i<m; i++)
- c[i + j * c_dim1] = (GFC_COMPLEX_17)0;
+ C_ARRAY_ELEM (i, j) = (GFC_COMPLEX_17)0;
/* Early exit if possible */
if (m == 0 || n == 0 || k == 0)
@@ -1509,20 +1555,20 @@ matmul_c17_avx512f (gfc_array_c17 * const restrict retarray,
for (i = ii; i < i5; i += 2)
{
t1[l - ll + 1 + ((i - ii + 1) << 8) - 257] =
- a[i + l * a_dim1];
+ A_ARRAY_ELEM (i, l);
t1[l - ll + 2 + ((i - ii + 1) << 8) - 257] =
- a[i + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i, l + 1);
t1[l - ll + 1 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + l * a_dim1];
+ A_ARRAY_ELEM (i + 1, l);
t1[l - ll + 2 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i + 1, l + 1);
}
if (uisec < isec)
{
t1[l - ll + 1 + (isec << 8) - 257] =
- a[ii + isec - 1 + l * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l);
t1[l - ll + 2 + (isec << 8) - 257] =
- a[ii + isec - 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l + 1);
}
}
if (ulsec < lsec)
@@ -1531,7 +1577,7 @@ matmul_c17_avx512f (gfc_array_c17 * const restrict retarray,
for (i = ii; i< i4; ++i)
{
t1[lsec + ((i - ii + 1) << 8) - 257] =
- a[i + (ll + lsec - 1) * a_dim1];
+ A_ARRAY_ELEM (i, ll + lsec - 1);
}
}
@@ -1542,100 +1588,100 @@ matmul_c17_avx512f (gfc_array_c17 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f22 = c[i + 1 + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f23 = c[i + 1 + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
- f24 = c[i + 1 + (j + 3) * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
- f32 = c[i + 2 + (j + 1) * c_dim1];
- f42 = c[i + 3 + (j + 1) * c_dim1];
- f33 = c[i + 2 + (j + 2) * c_dim1];
- f43 = c[i + 3 + (j + 2) * c_dim1];
- f34 = c[i + 2 + (j + 3) * c_dim1];
- f44 = c[i + 3 + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f22 = C_ARRAY_ELEM (i + 1, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f23 = C_ARRAY_ELEM (i + 1, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
+ f24 = C_ARRAY_ELEM (i + 1, j + 3);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
+ f32 = C_ARRAY_ELEM (i + 2, j + 1);
+ f42 = C_ARRAY_ELEM (i + 3, j + 1);
+ f33 = C_ARRAY_ELEM (i + 2, j + 2);
+ f43 = C_ARRAY_ELEM (i + 3, j + 2);
+ f34 = C_ARRAY_ELEM (i + 2, j + 3);
+ f44 = C_ARRAY_ELEM (i + 3, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f22 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f23 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f24 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f32 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f42 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f33 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f43 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f34 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f44 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + 1 + (j + 1) * c_dim1] = f22;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + 1 + (j + 2) * c_dim1] = f23;
- c[i + (j + 3) * c_dim1] = f14;
- c[i + 1 + (j + 3) * c_dim1] = f24;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
- c[i + 2 + (j + 1) * c_dim1] = f32;
- c[i + 3 + (j + 1) * c_dim1] = f42;
- c[i + 2 + (j + 2) * c_dim1] = f33;
- c[i + 3 + (j + 2) * c_dim1] = f43;
- c[i + 2 + (j + 3) * c_dim1] = f34;
- c[i + 3 + (j + 3) * c_dim1] = f44;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i + 1, j + 1) = f22;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i + 1, j + 2) = f23;
+ C_ARRAY_ELEM (i, j + 3) = f14;
+ C_ARRAY_ELEM (i + 1, j + 3) = f24;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
+ C_ARRAY_ELEM (i + 2, j + 1) = f32;
+ C_ARRAY_ELEM (i + 3, j + 1) = f42;
+ C_ARRAY_ELEM (i + 2, j + 2) = f33;
+ C_ARRAY_ELEM (i + 3, j + 2) = f43;
+ C_ARRAY_ELEM (i + 2, j + 3) = f34;
+ C_ARRAY_ELEM (i + 3, j + 3) = f44;
}
if (uisec < isec)
{
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 1) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 2) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 3) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + (j + 3) * c_dim1] = f14;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i, j + 3) = f14;
}
}
}
@@ -1647,38 +1693,38 @@ matmul_c17_avx512f (gfc_array_c17 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
}
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
+ C_ARRAY_ELEM (i, j) = f11;
}
}
}
@@ -1687,6 +1733,9 @@ matmul_c17_avx512f (gfc_array_c17 * const restrict retarray,
}
free(t1);
return;
+#undef A_ARRAY_ELEM
+#undef B_ARRAY_ELEM
+#undef C_ARRAY_ELEM
}
else if (rxstride == 1 && aystride == 1 && bxstride == 1)
{
@@ -1699,11 +1748,11 @@ matmul_c17_avx512f (gfc_array_c17 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_17) 0;
for (n = 0; n < count; n++)
s += abase_x[n] * bbase_y[n];
@@ -1718,11 +1767,11 @@ matmul_c17_avx512f (gfc_array_c17 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_17) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n];
- dest[y*rystride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n) * bbase_y[n];
+ ARRAY_ELEM_AT_OFFSET (dest, y * rystride_bytes) = s;
}
}
}
@@ -1733,26 +1782,27 @@ matmul_c17_avx512f (gfc_array_c17 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_17) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n*bxstride];
- dest[y*rxstride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
}
}
else if (axstride < aystride)
{
for (y = 0; y < ycount; y++)
for (x = 0; x < xcount; x++)
- dest[x*rxstride + y*rystride] = (GFC_COMPLEX_17)0;
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y) = (GFC_COMPLEX_17)0;
for (y = 0; y < ycount; y++)
for (n = 0; n < count; n++)
for (x = 0; x < xcount; x++)
/* dest[x,y] += a[x,n] * b[n,y] */
- dest[x*rxstride + y*rystride] +=
- abase[x*axstride + n*aystride] *
- bbase[n*bxstride + y*bystride];
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y)
+ += GFC_DESCRIPTOR2_ELEM (a, x, n)
+ * GFC_DESCRIPTOR2_ELEM (b, n, y);
}
else
{
@@ -1763,15 +1813,16 @@ matmul_c17_avx512f (gfc_array_c17 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_17) 0;
for (n = 0; n < count; n++)
- s += abase_x[n*aystride] * bbase_y[n*bxstride];
- dest_y[x*rxstride] = s;
+ s += ARRAY_ELEM_AT_OFFSET (abase_x, n * aystride_bytes)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ ARRAY_ELEM_AT_OFFSET (dest_y, x * rxstride_bytes) = s;
}
}
}
@@ -1812,7 +1863,8 @@ matmul_c17_vanilla (gfc_array_c17 * const restrict retarray,
index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
index_type x, y, n, count, xcount, ycount;
- index_type aystride_bytes, bystride_bytes, rystride_bytes;
+ index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
+ rxstride_bytes, rystride_bytes;
assert (GFC_DESCRIPTOR_RANK (a) == 2
|| GFC_DESCRIPTOR_RANK (b) == 2);
@@ -1901,12 +1953,13 @@ matmul_c17_vanilla (gfc_array_c17 * const restrict retarray,
either as a row or a column matrix. We want both cases to
work. */
rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
+ rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
}
else
{
rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
+ rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
}
@@ -1915,6 +1968,7 @@ matmul_c17_vanilla (gfc_array_c17 * const restrict retarray,
/* Treat it as a a row matrix A[1,count]. */
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = 1;
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = sizeof (GFC_COMPLEX_17);
xcount = 1;
@@ -1924,6 +1978,7 @@ matmul_c17_vanilla (gfc_array_c17 * const restrict retarray,
{
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = GFC_DESCRIPTOR_STRIDE(a,1);
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,1);
count = GFC_DESCRIPTOR_EXTENT(a,1);
@@ -1942,17 +1997,20 @@ matmul_c17_vanilla (gfc_array_c17 * const restrict retarray,
{
/* Treat it as a column matrix B[count,1] */
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
/* bystride should never be used for 1-dimensional b.
The value is only used for calculation of the
memory by the buffer. */
bystride = 256;
+ bystride_bytes = 99999999;
ycount = 1;
}
else
{
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
bystride = GFC_DESCRIPTOR_STRIDE(b,1);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
bystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,1);
ycount = GFC_DESCRIPTOR_EXTENT(b,1);
}
@@ -2012,12 +2070,11 @@ matmul_c17_vanilla (gfc_array_c17 * const restrict retarray,
from netlib.org, translated to C, and modified for matmul.m4. */
- const GFC_COMPLEX_17 *a, *b;
GFC_COMPLEX_17 *c;
const index_type m = xcount, n = ycount, k = count;
/* System generated locals */
- index_type a_dim1, b_dim1, c_dim1,
+ index_type a_dim1, b_dim1,
i1, i2, i3, i4, i5, i6;
/* Local variables */
@@ -2027,19 +2084,25 @@ matmul_c17_vanilla (gfc_array_c17 * const restrict retarray,
index_type isec, jsec, lsec, uisec, ujsec, ulsec;
GFC_COMPLEX_17 *t1;
- a = abase;
- b = bbase;
c = retarray->base_addr;
/* Parameter adjustments */
- c_dim1 = rystride;
a_dim1 = aystride;
b_dim1 = bystride;
- /* Empty c first. */
+#define A_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (abase, (i) * sizeof (GFC_COMPLEX_17) + (j) * aystride_bytes))
+
+#define B_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (bbase, (i) * sizeof (GFC_COMPLEX_17) + (j) * bystride_bytes))
+
+#define C_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (c, (i) * sizeof (GFC_COMPLEX_17) + (j) * rystride_bytes))
+
+ /* Empty result first. */
for (j=0; j<n; j++)
for (i=0; i<m; i++)
- c[i + j * c_dim1] = (GFC_COMPLEX_17)0;
+ C_ARRAY_ELEM (i, j) = (GFC_COMPLEX_17)0;
/* Early exit if possible */
if (m == 0 || n == 0 || k == 0)
@@ -2091,20 +2154,20 @@ matmul_c17_vanilla (gfc_array_c17 * const restrict retarray,
for (i = ii; i < i5; i += 2)
{
t1[l - ll + 1 + ((i - ii + 1) << 8) - 257] =
- a[i + l * a_dim1];
+ A_ARRAY_ELEM (i, l);
t1[l - ll + 2 + ((i - ii + 1) << 8) - 257] =
- a[i + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i, l + 1);
t1[l - ll + 1 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + l * a_dim1];
+ A_ARRAY_ELEM (i + 1, l);
t1[l - ll + 2 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i + 1, l + 1);
}
if (uisec < isec)
{
t1[l - ll + 1 + (isec << 8) - 257] =
- a[ii + isec - 1 + l * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l);
t1[l - ll + 2 + (isec << 8) - 257] =
- a[ii + isec - 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l + 1);
}
}
if (ulsec < lsec)
@@ -2113,7 +2176,7 @@ matmul_c17_vanilla (gfc_array_c17 * const restrict retarray,
for (i = ii; i< i4; ++i)
{
t1[lsec + ((i - ii + 1) << 8) - 257] =
- a[i + (ll + lsec - 1) * a_dim1];
+ A_ARRAY_ELEM (i, ll + lsec - 1);
}
}
@@ -2124,100 +2187,100 @@ matmul_c17_vanilla (gfc_array_c17 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f22 = c[i + 1 + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f23 = c[i + 1 + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
- f24 = c[i + 1 + (j + 3) * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
- f32 = c[i + 2 + (j + 1) * c_dim1];
- f42 = c[i + 3 + (j + 1) * c_dim1];
- f33 = c[i + 2 + (j + 2) * c_dim1];
- f43 = c[i + 3 + (j + 2) * c_dim1];
- f34 = c[i + 2 + (j + 3) * c_dim1];
- f44 = c[i + 3 + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f22 = C_ARRAY_ELEM (i + 1, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f23 = C_ARRAY_ELEM (i + 1, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
+ f24 = C_ARRAY_ELEM (i + 1, j + 3);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
+ f32 = C_ARRAY_ELEM (i + 2, j + 1);
+ f42 = C_ARRAY_ELEM (i + 3, j + 1);
+ f33 = C_ARRAY_ELEM (i + 2, j + 2);
+ f43 = C_ARRAY_ELEM (i + 3, j + 2);
+ f34 = C_ARRAY_ELEM (i + 2, j + 3);
+ f44 = C_ARRAY_ELEM (i + 3, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f22 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f23 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f24 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f32 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f42 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f33 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f43 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f34 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f44 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + 1 + (j + 1) * c_dim1] = f22;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + 1 + (j + 2) * c_dim1] = f23;
- c[i + (j + 3) * c_dim1] = f14;
- c[i + 1 + (j + 3) * c_dim1] = f24;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
- c[i + 2 + (j + 1) * c_dim1] = f32;
- c[i + 3 + (j + 1) * c_dim1] = f42;
- c[i + 2 + (j + 2) * c_dim1] = f33;
- c[i + 3 + (j + 2) * c_dim1] = f43;
- c[i + 2 + (j + 3) * c_dim1] = f34;
- c[i + 3 + (j + 3) * c_dim1] = f44;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i + 1, j + 1) = f22;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i + 1, j + 2) = f23;
+ C_ARRAY_ELEM (i, j + 3) = f14;
+ C_ARRAY_ELEM (i + 1, j + 3) = f24;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
+ C_ARRAY_ELEM (i + 2, j + 1) = f32;
+ C_ARRAY_ELEM (i + 3, j + 1) = f42;
+ C_ARRAY_ELEM (i + 2, j + 2) = f33;
+ C_ARRAY_ELEM (i + 3, j + 2) = f43;
+ C_ARRAY_ELEM (i + 2, j + 3) = f34;
+ C_ARRAY_ELEM (i + 3, j + 3) = f44;
}
if (uisec < isec)
{
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 1) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 2) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 3) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + (j + 3) * c_dim1] = f14;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i, j + 3) = f14;
}
}
}
@@ -2229,38 +2292,38 @@ matmul_c17_vanilla (gfc_array_c17 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
}
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
+ C_ARRAY_ELEM (i, j) = f11;
}
}
}
@@ -2269,6 +2332,9 @@ matmul_c17_vanilla (gfc_array_c17 * const restrict retarray,
}
free(t1);
return;
+#undef A_ARRAY_ELEM
+#undef B_ARRAY_ELEM
+#undef C_ARRAY_ELEM
}
else if (rxstride == 1 && aystride == 1 && bxstride == 1)
{
@@ -2281,11 +2347,11 @@ matmul_c17_vanilla (gfc_array_c17 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_17) 0;
for (n = 0; n < count; n++)
s += abase_x[n] * bbase_y[n];
@@ -2300,11 +2366,11 @@ matmul_c17_vanilla (gfc_array_c17 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_17) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n];
- dest[y*rystride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n) * bbase_y[n];
+ ARRAY_ELEM_AT_OFFSET (dest, y * rystride_bytes) = s;
}
}
}
@@ -2315,26 +2381,27 @@ matmul_c17_vanilla (gfc_array_c17 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_17) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n*bxstride];
- dest[y*rxstride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
}
}
else if (axstride < aystride)
{
for (y = 0; y < ycount; y++)
for (x = 0; x < xcount; x++)
- dest[x*rxstride + y*rystride] = (GFC_COMPLEX_17)0;
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y) = (GFC_COMPLEX_17)0;
for (y = 0; y < ycount; y++)
for (n = 0; n < count; n++)
for (x = 0; x < xcount; x++)
/* dest[x,y] += a[x,n] * b[n,y] */
- dest[x*rxstride + y*rystride] +=
- abase[x*axstride + n*aystride] *
- bbase[n*bxstride + y*bystride];
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y)
+ += GFC_DESCRIPTOR2_ELEM (a, x, n)
+ * GFC_DESCRIPTOR2_ELEM (b, n, y);
}
else
{
@@ -2345,15 +2412,16 @@ matmul_c17_vanilla (gfc_array_c17 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_17) 0;
for (n = 0; n < count; n++)
- s += abase_x[n*aystride] * bbase_y[n*bxstride];
- dest_y[x*rxstride] = s;
+ s += ARRAY_ELEM_AT_OFFSET (abase_x, n * aystride_bytes)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ ARRAY_ELEM_AT_OFFSET (dest_y, x * rxstride_bytes) = s;
}
}
}
@@ -2453,7 +2521,8 @@ matmul_c17 (gfc_array_c17 * const restrict retarray,
index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
index_type x, y, n, count, xcount, ycount;
- index_type aystride_bytes, bystride_bytes, rystride_bytes;
+ index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
+ rxstride_bytes, rystride_bytes;
assert (GFC_DESCRIPTOR_RANK (a) == 2
|| GFC_DESCRIPTOR_RANK (b) == 2);
@@ -2542,12 +2611,13 @@ matmul_c17 (gfc_array_c17 * const restrict retarray,
either as a row or a column matrix. We want both cases to
work. */
rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
+ rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
}
else
{
rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
+ rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
}
@@ -2556,6 +2626,7 @@ matmul_c17 (gfc_array_c17 * const restrict retarray,
/* Treat it as a a row matrix A[1,count]. */
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = 1;
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = sizeof (GFC_COMPLEX_17);
xcount = 1;
@@ -2565,6 +2636,7 @@ matmul_c17 (gfc_array_c17 * const restrict retarray,
{
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = GFC_DESCRIPTOR_STRIDE(a,1);
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,1);
count = GFC_DESCRIPTOR_EXTENT(a,1);
@@ -2583,17 +2655,20 @@ matmul_c17 (gfc_array_c17 * const restrict retarray,
{
/* Treat it as a column matrix B[count,1] */
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
/* bystride should never be used for 1-dimensional b.
The value is only used for calculation of the
memory by the buffer. */
bystride = 256;
+ bystride_bytes = 99999999;
ycount = 1;
}
else
{
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
bystride = GFC_DESCRIPTOR_STRIDE(b,1);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
bystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,1);
ycount = GFC_DESCRIPTOR_EXTENT(b,1);
}
@@ -2653,12 +2728,11 @@ matmul_c17 (gfc_array_c17 * const restrict retarray,
from netlib.org, translated to C, and modified for matmul.m4. */
- const GFC_COMPLEX_17 *a, *b;
GFC_COMPLEX_17 *c;
const index_type m = xcount, n = ycount, k = count;
/* System generated locals */
- index_type a_dim1, b_dim1, c_dim1,
+ index_type a_dim1, b_dim1,
i1, i2, i3, i4, i5, i6;
/* Local variables */
@@ -2668,19 +2742,25 @@ matmul_c17 (gfc_array_c17 * const restrict retarray,
index_type isec, jsec, lsec, uisec, ujsec, ulsec;
GFC_COMPLEX_17 *t1;
- a = abase;
- b = bbase;
c = retarray->base_addr;
/* Parameter adjustments */
- c_dim1 = rystride;
a_dim1 = aystride;
b_dim1 = bystride;
- /* Empty c first. */
+#define A_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (abase, (i) * sizeof (GFC_COMPLEX_17) + (j) * aystride_bytes))
+
+#define B_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (bbase, (i) * sizeof (GFC_COMPLEX_17) + (j) * bystride_bytes))
+
+#define C_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (c, (i) * sizeof (GFC_COMPLEX_17) + (j) * rystride_bytes))
+
+ /* Empty result first. */
for (j=0; j<n; j++)
for (i=0; i<m; i++)
- c[i + j * c_dim1] = (GFC_COMPLEX_17)0;
+ C_ARRAY_ELEM (i, j) = (GFC_COMPLEX_17)0;
/* Early exit if possible */
if (m == 0 || n == 0 || k == 0)
@@ -2732,20 +2812,20 @@ matmul_c17 (gfc_array_c17 * const restrict retarray,
for (i = ii; i < i5; i += 2)
{
t1[l - ll + 1 + ((i - ii + 1) << 8) - 257] =
- a[i + l * a_dim1];
+ A_ARRAY_ELEM (i, l);
t1[l - ll + 2 + ((i - ii + 1) << 8) - 257] =
- a[i + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i, l + 1);
t1[l - ll + 1 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + l * a_dim1];
+ A_ARRAY_ELEM (i + 1, l);
t1[l - ll + 2 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i + 1, l + 1);
}
if (uisec < isec)
{
t1[l - ll + 1 + (isec << 8) - 257] =
- a[ii + isec - 1 + l * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l);
t1[l - ll + 2 + (isec << 8) - 257] =
- a[ii + isec - 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l + 1);
}
}
if (ulsec < lsec)
@@ -2754,7 +2834,7 @@ matmul_c17 (gfc_array_c17 * const restrict retarray,
for (i = ii; i< i4; ++i)
{
t1[lsec + ((i - ii + 1) << 8) - 257] =
- a[i + (ll + lsec - 1) * a_dim1];
+ A_ARRAY_ELEM (i, ll + lsec - 1);
}
}
@@ -2765,100 +2845,100 @@ matmul_c17 (gfc_array_c17 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f22 = c[i + 1 + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f23 = c[i + 1 + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
- f24 = c[i + 1 + (j + 3) * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
- f32 = c[i + 2 + (j + 1) * c_dim1];
- f42 = c[i + 3 + (j + 1) * c_dim1];
- f33 = c[i + 2 + (j + 2) * c_dim1];
- f43 = c[i + 3 + (j + 2) * c_dim1];
- f34 = c[i + 2 + (j + 3) * c_dim1];
- f44 = c[i + 3 + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f22 = C_ARRAY_ELEM (i + 1, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f23 = C_ARRAY_ELEM (i + 1, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
+ f24 = C_ARRAY_ELEM (i + 1, j + 3);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
+ f32 = C_ARRAY_ELEM (i + 2, j + 1);
+ f42 = C_ARRAY_ELEM (i + 3, j + 1);
+ f33 = C_ARRAY_ELEM (i + 2, j + 2);
+ f43 = C_ARRAY_ELEM (i + 3, j + 2);
+ f34 = C_ARRAY_ELEM (i + 2, j + 3);
+ f44 = C_ARRAY_ELEM (i + 3, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f22 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f23 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f24 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f32 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f42 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f33 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f43 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f34 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f44 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + 1 + (j + 1) * c_dim1] = f22;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + 1 + (j + 2) * c_dim1] = f23;
- c[i + (j + 3) * c_dim1] = f14;
- c[i + 1 + (j + 3) * c_dim1] = f24;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
- c[i + 2 + (j + 1) * c_dim1] = f32;
- c[i + 3 + (j + 1) * c_dim1] = f42;
- c[i + 2 + (j + 2) * c_dim1] = f33;
- c[i + 3 + (j + 2) * c_dim1] = f43;
- c[i + 2 + (j + 3) * c_dim1] = f34;
- c[i + 3 + (j + 3) * c_dim1] = f44;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i + 1, j + 1) = f22;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i + 1, j + 2) = f23;
+ C_ARRAY_ELEM (i, j + 3) = f14;
+ C_ARRAY_ELEM (i + 1, j + 3) = f24;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
+ C_ARRAY_ELEM (i + 2, j + 1) = f32;
+ C_ARRAY_ELEM (i + 3, j + 1) = f42;
+ C_ARRAY_ELEM (i + 2, j + 2) = f33;
+ C_ARRAY_ELEM (i + 3, j + 2) = f43;
+ C_ARRAY_ELEM (i + 2, j + 3) = f34;
+ C_ARRAY_ELEM (i + 3, j + 3) = f44;
}
if (uisec < isec)
{
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 1) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 2) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 3) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + (j + 3) * c_dim1] = f14;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i, j + 3) = f14;
}
}
}
@@ -2870,38 +2950,38 @@ matmul_c17 (gfc_array_c17 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
}
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
+ C_ARRAY_ELEM (i, j) = f11;
}
}
}
@@ -2910,6 +2990,9 @@ matmul_c17 (gfc_array_c17 * const restrict retarray,
}
free(t1);
return;
+#undef A_ARRAY_ELEM
+#undef B_ARRAY_ELEM
+#undef C_ARRAY_ELEM
}
else if (rxstride == 1 && aystride == 1 && bxstride == 1)
{
@@ -2922,11 +3005,11 @@ matmul_c17 (gfc_array_c17 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_17) 0;
for (n = 0; n < count; n++)
s += abase_x[n] * bbase_y[n];
@@ -2941,11 +3024,11 @@ matmul_c17 (gfc_array_c17 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_17) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n];
- dest[y*rystride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n) * bbase_y[n];
+ ARRAY_ELEM_AT_OFFSET (dest, y * rystride_bytes) = s;
}
}
}
@@ -2956,26 +3039,27 @@ matmul_c17 (gfc_array_c17 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_17) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n*bxstride];
- dest[y*rxstride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
}
}
else if (axstride < aystride)
{
for (y = 0; y < ycount; y++)
for (x = 0; x < xcount; x++)
- dest[x*rxstride + y*rystride] = (GFC_COMPLEX_17)0;
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y) = (GFC_COMPLEX_17)0;
for (y = 0; y < ycount; y++)
for (n = 0; n < count; n++)
for (x = 0; x < xcount; x++)
/* dest[x,y] += a[x,n] * b[n,y] */
- dest[x*rxstride + y*rystride] +=
- abase[x*axstride + n*aystride] *
- bbase[n*bxstride + y*bystride];
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y)
+ += GFC_DESCRIPTOR2_ELEM (a, x, n)
+ * GFC_DESCRIPTOR2_ELEM (b, n, y);
}
else
{
@@ -2986,15 +3070,16 @@ matmul_c17 (gfc_array_c17 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_17) 0;
for (n = 0; n < count; n++)
- s += abase_x[n*aystride] * bbase_y[n*bxstride];
- dest_y[x*rxstride] = s;
+ s += ARRAY_ELEM_AT_OFFSET (abase_x, n * aystride_bytes)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ ARRAY_ELEM_AT_OFFSET (dest_y, x * rxstride_bytes) = s;
}
}
}
diff --git a/libgfortran/generated/matmul_c4.c b/libgfortran/generated/matmul_c4.c
index f9b65bd150aa..afb9f4f1473d 100644
--- a/libgfortran/generated/matmul_c4.c
+++ b/libgfortran/generated/matmul_c4.c
@@ -94,7 +94,8 @@ matmul_c4_avx (gfc_array_c4 * const restrict retarray,
index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
index_type x, y, n, count, xcount, ycount;
- index_type aystride_bytes, bystride_bytes, rystride_bytes;
+ index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
+ rxstride_bytes, rystride_bytes;
assert (GFC_DESCRIPTOR_RANK (a) == 2
|| GFC_DESCRIPTOR_RANK (b) == 2);
@@ -183,12 +184,13 @@ matmul_c4_avx (gfc_array_c4 * const restrict retarray,
either as a row or a column matrix. We want both cases to
work. */
rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
+ rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
}
else
{
rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
+ rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
}
@@ -197,6 +199,7 @@ matmul_c4_avx (gfc_array_c4 * const restrict retarray,
/* Treat it as a a row matrix A[1,count]. */
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = 1;
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = sizeof (GFC_COMPLEX_4);
xcount = 1;
@@ -206,6 +209,7 @@ matmul_c4_avx (gfc_array_c4 * const restrict retarray,
{
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = GFC_DESCRIPTOR_STRIDE(a,1);
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,1);
count = GFC_DESCRIPTOR_EXTENT(a,1);
@@ -224,17 +228,20 @@ matmul_c4_avx (gfc_array_c4 * const restrict retarray,
{
/* Treat it as a column matrix B[count,1] */
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
/* bystride should never be used for 1-dimensional b.
The value is only used for calculation of the
memory by the buffer. */
bystride = 256;
+ bystride_bytes = 99999999;
ycount = 1;
}
else
{
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
bystride = GFC_DESCRIPTOR_STRIDE(b,1);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
bystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,1);
ycount = GFC_DESCRIPTOR_EXTENT(b,1);
}
@@ -294,12 +301,11 @@ matmul_c4_avx (gfc_array_c4 * const restrict retarray,
from netlib.org, translated to C, and modified for matmul.m4. */
- const GFC_COMPLEX_4 *a, *b;
GFC_COMPLEX_4 *c;
const index_type m = xcount, n = ycount, k = count;
/* System generated locals */
- index_type a_dim1, b_dim1, c_dim1,
+ index_type a_dim1, b_dim1,
i1, i2, i3, i4, i5, i6;
/* Local variables */
@@ -309,19 +315,25 @@ matmul_c4_avx (gfc_array_c4 * const restrict retarray,
index_type isec, jsec, lsec, uisec, ujsec, ulsec;
GFC_COMPLEX_4 *t1;
- a = abase;
- b = bbase;
c = retarray->base_addr;
/* Parameter adjustments */
- c_dim1 = rystride;
a_dim1 = aystride;
b_dim1 = bystride;
- /* Empty c first. */
+#define A_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (abase, (i) * sizeof (GFC_COMPLEX_4) + (j) * aystride_bytes))
+
+#define B_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (bbase, (i) * sizeof (GFC_COMPLEX_4) + (j) * bystride_bytes))
+
+#define C_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (c, (i) * sizeof (GFC_COMPLEX_4) + (j) * rystride_bytes))
+
+ /* Empty result first. */
for (j=0; j<n; j++)
for (i=0; i<m; i++)
- c[i + j * c_dim1] = (GFC_COMPLEX_4)0;
+ C_ARRAY_ELEM (i, j) = (GFC_COMPLEX_4)0;
/* Early exit if possible */
if (m == 0 || n == 0 || k == 0)
@@ -373,20 +385,20 @@ matmul_c4_avx (gfc_array_c4 * const restrict retarray,
for (i = ii; i < i5; i += 2)
{
t1[l - ll + 1 + ((i - ii + 1) << 8) - 257] =
- a[i + l * a_dim1];
+ A_ARRAY_ELEM (i, l);
t1[l - ll + 2 + ((i - ii + 1) << 8) - 257] =
- a[i + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i, l + 1);
t1[l - ll + 1 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + l * a_dim1];
+ A_ARRAY_ELEM (i + 1, l);
t1[l - ll + 2 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i + 1, l + 1);
}
if (uisec < isec)
{
t1[l - ll + 1 + (isec << 8) - 257] =
- a[ii + isec - 1 + l * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l);
t1[l - ll + 2 + (isec << 8) - 257] =
- a[ii + isec - 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l + 1);
}
}
if (ulsec < lsec)
@@ -395,7 +407,7 @@ matmul_c4_avx (gfc_array_c4 * const restrict retarray,
for (i = ii; i< i4; ++i)
{
t1[lsec + ((i - ii + 1) << 8) - 257] =
- a[i + (ll + lsec - 1) * a_dim1];
+ A_ARRAY_ELEM (i, ll + lsec - 1);
}
}
@@ -406,100 +418,100 @@ matmul_c4_avx (gfc_array_c4 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f22 = c[i + 1 + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f23 = c[i + 1 + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
- f24 = c[i + 1 + (j + 3) * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
- f32 = c[i + 2 + (j + 1) * c_dim1];
- f42 = c[i + 3 + (j + 1) * c_dim1];
- f33 = c[i + 2 + (j + 2) * c_dim1];
- f43 = c[i + 3 + (j + 2) * c_dim1];
- f34 = c[i + 2 + (j + 3) * c_dim1];
- f44 = c[i + 3 + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f22 = C_ARRAY_ELEM (i + 1, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f23 = C_ARRAY_ELEM (i + 1, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
+ f24 = C_ARRAY_ELEM (i + 1, j + 3);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
+ f32 = C_ARRAY_ELEM (i + 2, j + 1);
+ f42 = C_ARRAY_ELEM (i + 3, j + 1);
+ f33 = C_ARRAY_ELEM (i + 2, j + 2);
+ f43 = C_ARRAY_ELEM (i + 3, j + 2);
+ f34 = C_ARRAY_ELEM (i + 2, j + 3);
+ f44 = C_ARRAY_ELEM (i + 3, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f22 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f23 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f24 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f32 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f42 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f33 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f43 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f34 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f44 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + 1 + (j + 1) * c_dim1] = f22;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + 1 + (j + 2) * c_dim1] = f23;
- c[i + (j + 3) * c_dim1] = f14;
- c[i + 1 + (j + 3) * c_dim1] = f24;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
- c[i + 2 + (j + 1) * c_dim1] = f32;
- c[i + 3 + (j + 1) * c_dim1] = f42;
- c[i + 2 + (j + 2) * c_dim1] = f33;
- c[i + 3 + (j + 2) * c_dim1] = f43;
- c[i + 2 + (j + 3) * c_dim1] = f34;
- c[i + 3 + (j + 3) * c_dim1] = f44;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i + 1, j + 1) = f22;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i + 1, j + 2) = f23;
+ C_ARRAY_ELEM (i, j + 3) = f14;
+ C_ARRAY_ELEM (i + 1, j + 3) = f24;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
+ C_ARRAY_ELEM (i + 2, j + 1) = f32;
+ C_ARRAY_ELEM (i + 3, j + 1) = f42;
+ C_ARRAY_ELEM (i + 2, j + 2) = f33;
+ C_ARRAY_ELEM (i + 3, j + 2) = f43;
+ C_ARRAY_ELEM (i + 2, j + 3) = f34;
+ C_ARRAY_ELEM (i + 3, j + 3) = f44;
}
if (uisec < isec)
{
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 1) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 2) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 3) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + (j + 3) * c_dim1] = f14;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i, j + 3) = f14;
}
}
}
@@ -511,38 +523,38 @@ matmul_c4_avx (gfc_array_c4 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
}
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
+ C_ARRAY_ELEM (i, j) = f11;
}
}
}
@@ -551,6 +563,9 @@ matmul_c4_avx (gfc_array_c4 * const restrict retarray,
}
free(t1);
return;
+#undef A_ARRAY_ELEM
+#undef B_ARRAY_ELEM
+#undef C_ARRAY_ELEM
}
else if (rxstride == 1 && aystride == 1 && bxstride == 1)
{
@@ -563,11 +578,11 @@ matmul_c4_avx (gfc_array_c4 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_4) 0;
for (n = 0; n < count; n++)
s += abase_x[n] * bbase_y[n];
@@ -582,11 +597,11 @@ matmul_c4_avx (gfc_array_c4 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_4) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n];
- dest[y*rystride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n) * bbase_y[n];
+ ARRAY_ELEM_AT_OFFSET (dest, y * rystride_bytes) = s;
}
}
}
@@ -597,26 +612,27 @@ matmul_c4_avx (gfc_array_c4 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_4) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n*bxstride];
- dest[y*rxstride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
}
}
else if (axstride < aystride)
{
for (y = 0; y < ycount; y++)
for (x = 0; x < xcount; x++)
- dest[x*rxstride + y*rystride] = (GFC_COMPLEX_4)0;
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y) = (GFC_COMPLEX_4)0;
for (y = 0; y < ycount; y++)
for (n = 0; n < count; n++)
for (x = 0; x < xcount; x++)
/* dest[x,y] += a[x,n] * b[n,y] */
- dest[x*rxstride + y*rystride] +=
- abase[x*axstride + n*aystride] *
- bbase[n*bxstride + y*bystride];
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y)
+ += GFC_DESCRIPTOR2_ELEM (a, x, n)
+ * GFC_DESCRIPTOR2_ELEM (b, n, y);
}
else
{
@@ -627,15 +643,16 @@ matmul_c4_avx (gfc_array_c4 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_4) 0;
for (n = 0; n < count; n++)
- s += abase_x[n*aystride] * bbase_y[n*bxstride];
- dest_y[x*rxstride] = s;
+ s += ARRAY_ELEM_AT_OFFSET (abase_x, n * aystride_bytes)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ ARRAY_ELEM_AT_OFFSET (dest_y, x * rxstride_bytes) = s;
}
}
}
@@ -662,7 +679,8 @@ matmul_c4_avx2 (gfc_array_c4 * const restrict retarray,
index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
index_type x, y, n, count, xcount, ycount;
- index_type aystride_bytes, bystride_bytes, rystride_bytes;
+ index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
+ rxstride_bytes, rystride_bytes;
assert (GFC_DESCRIPTOR_RANK (a) == 2
|| GFC_DESCRIPTOR_RANK (b) == 2);
@@ -751,12 +769,13 @@ matmul_c4_avx2 (gfc_array_c4 * const restrict retarray,
either as a row or a column matrix. We want both cases to
work. */
rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
+ rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
}
else
{
rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
+ rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
}
@@ -765,6 +784,7 @@ matmul_c4_avx2 (gfc_array_c4 * const restrict retarray,
/* Treat it as a a row matrix A[1,count]. */
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = 1;
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = sizeof (GFC_COMPLEX_4);
xcount = 1;
@@ -774,6 +794,7 @@ matmul_c4_avx2 (gfc_array_c4 * const restrict retarray,
{
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = GFC_DESCRIPTOR_STRIDE(a,1);
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,1);
count = GFC_DESCRIPTOR_EXTENT(a,1);
@@ -792,17 +813,20 @@ matmul_c4_avx2 (gfc_array_c4 * const restrict retarray,
{
/* Treat it as a column matrix B[count,1] */
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
/* bystride should never be used for 1-dimensional b.
The value is only used for calculation of the
memory by the buffer. */
bystride = 256;
+ bystride_bytes = 99999999;
ycount = 1;
}
else
{
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
bystride = GFC_DESCRIPTOR_STRIDE(b,1);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
bystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,1);
ycount = GFC_DESCRIPTOR_EXTENT(b,1);
}
@@ -862,12 +886,11 @@ matmul_c4_avx2 (gfc_array_c4 * const restrict retarray,
from netlib.org, translated to C, and modified for matmul.m4. */
- const GFC_COMPLEX_4 *a, *b;
GFC_COMPLEX_4 *c;
const index_type m = xcount, n = ycount, k = count;
/* System generated locals */
- index_type a_dim1, b_dim1, c_dim1,
+ index_type a_dim1, b_dim1,
i1, i2, i3, i4, i5, i6;
/* Local variables */
@@ -877,19 +900,25 @@ matmul_c4_avx2 (gfc_array_c4 * const restrict retarray,
index_type isec, jsec, lsec, uisec, ujsec, ulsec;
GFC_COMPLEX_4 *t1;
- a = abase;
- b = bbase;
c = retarray->base_addr;
/* Parameter adjustments */
- c_dim1 = rystride;
a_dim1 = aystride;
b_dim1 = bystride;
- /* Empty c first. */
+#define A_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (abase, (i) * sizeof (GFC_COMPLEX_4) + (j) * aystride_bytes))
+
+#define B_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (bbase, (i) * sizeof (GFC_COMPLEX_4) + (j) * bystride_bytes))
+
+#define C_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (c, (i) * sizeof (GFC_COMPLEX_4) + (j) * rystride_bytes))
+
+ /* Empty result first. */
for (j=0; j<n; j++)
for (i=0; i<m; i++)
- c[i + j * c_dim1] = (GFC_COMPLEX_4)0;
+ C_ARRAY_ELEM (i, j) = (GFC_COMPLEX_4)0;
/* Early exit if possible */
if (m == 0 || n == 0 || k == 0)
@@ -941,20 +970,20 @@ matmul_c4_avx2 (gfc_array_c4 * const restrict retarray,
for (i = ii; i < i5; i += 2)
{
t1[l - ll + 1 + ((i - ii + 1) << 8) - 257] =
- a[i + l * a_dim1];
+ A_ARRAY_ELEM (i, l);
t1[l - ll + 2 + ((i - ii + 1) << 8) - 257] =
- a[i + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i, l + 1);
t1[l - ll + 1 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + l * a_dim1];
+ A_ARRAY_ELEM (i + 1, l);
t1[l - ll + 2 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i + 1, l + 1);
}
if (uisec < isec)
{
t1[l - ll + 1 + (isec << 8) - 257] =
- a[ii + isec - 1 + l * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l);
t1[l - ll + 2 + (isec << 8) - 257] =
- a[ii + isec - 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l + 1);
}
}
if (ulsec < lsec)
@@ -963,7 +992,7 @@ matmul_c4_avx2 (gfc_array_c4 * const restrict retarray,
for (i = ii; i< i4; ++i)
{
t1[lsec + ((i - ii + 1) << 8) - 257] =
- a[i + (ll + lsec - 1) * a_dim1];
+ A_ARRAY_ELEM (i, ll + lsec - 1);
}
}
@@ -974,100 +1003,100 @@ matmul_c4_avx2 (gfc_array_c4 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f22 = c[i + 1 + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f23 = c[i + 1 + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
- f24 = c[i + 1 + (j + 3) * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
- f32 = c[i + 2 + (j + 1) * c_dim1];
- f42 = c[i + 3 + (j + 1) * c_dim1];
- f33 = c[i + 2 + (j + 2) * c_dim1];
- f43 = c[i + 3 + (j + 2) * c_dim1];
- f34 = c[i + 2 + (j + 3) * c_dim1];
- f44 = c[i + 3 + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f22 = C_ARRAY_ELEM (i + 1, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f23 = C_ARRAY_ELEM (i + 1, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
+ f24 = C_ARRAY_ELEM (i + 1, j + 3);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
+ f32 = C_ARRAY_ELEM (i + 2, j + 1);
+ f42 = C_ARRAY_ELEM (i + 3, j + 1);
+ f33 = C_ARRAY_ELEM (i + 2, j + 2);
+ f43 = C_ARRAY_ELEM (i + 3, j + 2);
+ f34 = C_ARRAY_ELEM (i + 2, j + 3);
+ f44 = C_ARRAY_ELEM (i + 3, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f22 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f23 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f24 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f32 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f42 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f33 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f43 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f34 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f44 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + 1 + (j + 1) * c_dim1] = f22;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + 1 + (j + 2) * c_dim1] = f23;
- c[i + (j + 3) * c_dim1] = f14;
- c[i + 1 + (j + 3) * c_dim1] = f24;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
- c[i + 2 + (j + 1) * c_dim1] = f32;
- c[i + 3 + (j + 1) * c_dim1] = f42;
- c[i + 2 + (j + 2) * c_dim1] = f33;
- c[i + 3 + (j + 2) * c_dim1] = f43;
- c[i + 2 + (j + 3) * c_dim1] = f34;
- c[i + 3 + (j + 3) * c_dim1] = f44;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i + 1, j + 1) = f22;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i + 1, j + 2) = f23;
+ C_ARRAY_ELEM (i, j + 3) = f14;
+ C_ARRAY_ELEM (i + 1, j + 3) = f24;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
+ C_ARRAY_ELEM (i + 2, j + 1) = f32;
+ C_ARRAY_ELEM (i + 3, j + 1) = f42;
+ C_ARRAY_ELEM (i + 2, j + 2) = f33;
+ C_ARRAY_ELEM (i + 3, j + 2) = f43;
+ C_ARRAY_ELEM (i + 2, j + 3) = f34;
+ C_ARRAY_ELEM (i + 3, j + 3) = f44;
}
if (uisec < isec)
{
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 1) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 2) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 3) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + (j + 3) * c_dim1] = f14;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i, j + 3) = f14;
}
}
}
@@ -1079,38 +1108,38 @@ matmul_c4_avx2 (gfc_array_c4 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
}
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
+ C_ARRAY_ELEM (i, j) = f11;
}
}
}
@@ -1119,6 +1148,9 @@ matmul_c4_avx2 (gfc_array_c4 * const restrict retarray,
}
free(t1);
return;
+#undef A_ARRAY_ELEM
+#undef B_ARRAY_ELEM
+#undef C_ARRAY_ELEM
}
else if (rxstride == 1 && aystride == 1 && bxstride == 1)
{
@@ -1131,11 +1163,11 @@ matmul_c4_avx2 (gfc_array_c4 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_4) 0;
for (n = 0; n < count; n++)
s += abase_x[n] * bbase_y[n];
@@ -1150,11 +1182,11 @@ matmul_c4_avx2 (gfc_array_c4 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_4) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n];
- dest[y*rystride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n) * bbase_y[n];
+ ARRAY_ELEM_AT_OFFSET (dest, y * rystride_bytes) = s;
}
}
}
@@ -1165,26 +1197,27 @@ matmul_c4_avx2 (gfc_array_c4 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_4) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n*bxstride];
- dest[y*rxstride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
}
}
else if (axstride < aystride)
{
for (y = 0; y < ycount; y++)
for (x = 0; x < xcount; x++)
- dest[x*rxstride + y*rystride] = (GFC_COMPLEX_4)0;
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y) = (GFC_COMPLEX_4)0;
for (y = 0; y < ycount; y++)
for (n = 0; n < count; n++)
for (x = 0; x < xcount; x++)
/* dest[x,y] += a[x,n] * b[n,y] */
- dest[x*rxstride + y*rystride] +=
- abase[x*axstride + n*aystride] *
- bbase[n*bxstride + y*bystride];
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y)
+ += GFC_DESCRIPTOR2_ELEM (a, x, n)
+ * GFC_DESCRIPTOR2_ELEM (b, n, y);
}
else
{
@@ -1195,15 +1228,16 @@ matmul_c4_avx2 (gfc_array_c4 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_4) 0;
for (n = 0; n < count; n++)
- s += abase_x[n*aystride] * bbase_y[n*bxstride];
- dest_y[x*rxstride] = s;
+ s += ARRAY_ELEM_AT_OFFSET (abase_x, n * aystride_bytes)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ ARRAY_ELEM_AT_OFFSET (dest_y, x * rxstride_bytes) = s;
}
}
}
@@ -1230,7 +1264,8 @@ matmul_c4_avx512f (gfc_array_c4 * const restrict retarray,
index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
index_type x, y, n, count, xcount, ycount;
- index_type aystride_bytes, bystride_bytes, rystride_bytes;
+ index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
+ rxstride_bytes, rystride_bytes;
assert (GFC_DESCRIPTOR_RANK (a) == 2
|| GFC_DESCRIPTOR_RANK (b) == 2);
@@ -1319,12 +1354,13 @@ matmul_c4_avx512f (gfc_array_c4 * const restrict retarray,
either as a row or a column matrix. We want both cases to
work. */
rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
+ rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
}
else
{
rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
+ rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
}
@@ -1333,6 +1369,7 @@ matmul_c4_avx512f (gfc_array_c4 * const restrict retarray,
/* Treat it as a a row matrix A[1,count]. */
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = 1;
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = sizeof (GFC_COMPLEX_4);
xcount = 1;
@@ -1342,6 +1379,7 @@ matmul_c4_avx512f (gfc_array_c4 * const restrict retarray,
{
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = GFC_DESCRIPTOR_STRIDE(a,1);
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,1);
count = GFC_DESCRIPTOR_EXTENT(a,1);
@@ -1360,17 +1398,20 @@ matmul_c4_avx512f (gfc_array_c4 * const restrict retarray,
{
/* Treat it as a column matrix B[count,1] */
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
/* bystride should never be used for 1-dimensional b.
The value is only used for calculation of the
memory by the buffer. */
bystride = 256;
+ bystride_bytes = 99999999;
ycount = 1;
}
else
{
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
bystride = GFC_DESCRIPTOR_STRIDE(b,1);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
bystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,1);
ycount = GFC_DESCRIPTOR_EXTENT(b,1);
}
@@ -1430,12 +1471,11 @@ matmul_c4_avx512f (gfc_array_c4 * const restrict retarray,
from netlib.org, translated to C, and modified for matmul.m4. */
- const GFC_COMPLEX_4 *a, *b;
GFC_COMPLEX_4 *c;
const index_type m = xcount, n = ycount, k = count;
/* System generated locals */
- index_type a_dim1, b_dim1, c_dim1,
+ index_type a_dim1, b_dim1,
i1, i2, i3, i4, i5, i6;
/* Local variables */
@@ -1445,19 +1485,25 @@ matmul_c4_avx512f (gfc_array_c4 * const restrict retarray,
index_type isec, jsec, lsec, uisec, ujsec, ulsec;
GFC_COMPLEX_4 *t1;
- a = abase;
- b = bbase;
c = retarray->base_addr;
/* Parameter adjustments */
- c_dim1 = rystride;
a_dim1 = aystride;
b_dim1 = bystride;
- /* Empty c first. */
+#define A_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (abase, (i) * sizeof (GFC_COMPLEX_4) + (j) * aystride_bytes))
+
+#define B_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (bbase, (i) * sizeof (GFC_COMPLEX_4) + (j) * bystride_bytes))
+
+#define C_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (c, (i) * sizeof (GFC_COMPLEX_4) + (j) * rystride_bytes))
+
+ /* Empty result first. */
for (j=0; j<n; j++)
for (i=0; i<m; i++)
- c[i + j * c_dim1] = (GFC_COMPLEX_4)0;
+ C_ARRAY_ELEM (i, j) = (GFC_COMPLEX_4)0;
/* Early exit if possible */
if (m == 0 || n == 0 || k == 0)
@@ -1509,20 +1555,20 @@ matmul_c4_avx512f (gfc_array_c4 * const restrict retarray,
for (i = ii; i < i5; i += 2)
{
t1[l - ll + 1 + ((i - ii + 1) << 8) - 257] =
- a[i + l * a_dim1];
+ A_ARRAY_ELEM (i, l);
t1[l - ll + 2 + ((i - ii + 1) << 8) - 257] =
- a[i + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i, l + 1);
t1[l - ll + 1 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + l * a_dim1];
+ A_ARRAY_ELEM (i + 1, l);
t1[l - ll + 2 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i + 1, l + 1);
}
if (uisec < isec)
{
t1[l - ll + 1 + (isec << 8) - 257] =
- a[ii + isec - 1 + l * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l);
t1[l - ll + 2 + (isec << 8) - 257] =
- a[ii + isec - 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l + 1);
}
}
if (ulsec < lsec)
@@ -1531,7 +1577,7 @@ matmul_c4_avx512f (gfc_array_c4 * const restrict retarray,
for (i = ii; i< i4; ++i)
{
t1[lsec + ((i - ii + 1) << 8) - 257] =
- a[i + (ll + lsec - 1) * a_dim1];
+ A_ARRAY_ELEM (i, ll + lsec - 1);
}
}
@@ -1542,100 +1588,100 @@ matmul_c4_avx512f (gfc_array_c4 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f22 = c[i + 1 + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f23 = c[i + 1 + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
- f24 = c[i + 1 + (j + 3) * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
- f32 = c[i + 2 + (j + 1) * c_dim1];
- f42 = c[i + 3 + (j + 1) * c_dim1];
- f33 = c[i + 2 + (j + 2) * c_dim1];
- f43 = c[i + 3 + (j + 2) * c_dim1];
- f34 = c[i + 2 + (j + 3) * c_dim1];
- f44 = c[i + 3 + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f22 = C_ARRAY_ELEM (i + 1, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f23 = C_ARRAY_ELEM (i + 1, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
+ f24 = C_ARRAY_ELEM (i + 1, j + 3);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
+ f32 = C_ARRAY_ELEM (i + 2, j + 1);
+ f42 = C_ARRAY_ELEM (i + 3, j + 1);
+ f33 = C_ARRAY_ELEM (i + 2, j + 2);
+ f43 = C_ARRAY_ELEM (i + 3, j + 2);
+ f34 = C_ARRAY_ELEM (i + 2, j + 3);
+ f44 = C_ARRAY_ELEM (i + 3, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f22 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f23 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f24 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f32 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f42 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f33 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f43 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f34 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f44 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + 1 + (j + 1) * c_dim1] = f22;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + 1 + (j + 2) * c_dim1] = f23;
- c[i + (j + 3) * c_dim1] = f14;
- c[i + 1 + (j + 3) * c_dim1] = f24;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
- c[i + 2 + (j + 1) * c_dim1] = f32;
- c[i + 3 + (j + 1) * c_dim1] = f42;
- c[i + 2 + (j + 2) * c_dim1] = f33;
- c[i + 3 + (j + 2) * c_dim1] = f43;
- c[i + 2 + (j + 3) * c_dim1] = f34;
- c[i + 3 + (j + 3) * c_dim1] = f44;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i + 1, j + 1) = f22;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i + 1, j + 2) = f23;
+ C_ARRAY_ELEM (i, j + 3) = f14;
+ C_ARRAY_ELEM (i + 1, j + 3) = f24;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
+ C_ARRAY_ELEM (i + 2, j + 1) = f32;
+ C_ARRAY_ELEM (i + 3, j + 1) = f42;
+ C_ARRAY_ELEM (i + 2, j + 2) = f33;
+ C_ARRAY_ELEM (i + 3, j + 2) = f43;
+ C_ARRAY_ELEM (i + 2, j + 3) = f34;
+ C_ARRAY_ELEM (i + 3, j + 3) = f44;
}
if (uisec < isec)
{
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 1) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 2) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 3) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + (j + 3) * c_dim1] = f14;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i, j + 3) = f14;
}
}
}
@@ -1647,38 +1693,38 @@ matmul_c4_avx512f (gfc_array_c4 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
}
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
+ C_ARRAY_ELEM (i, j) = f11;
}
}
}
@@ -1687,6 +1733,9 @@ matmul_c4_avx512f (gfc_array_c4 * const restrict retarray,
}
free(t1);
return;
+#undef A_ARRAY_ELEM
+#undef B_ARRAY_ELEM
+#undef C_ARRAY_ELEM
}
else if (rxstride == 1 && aystride == 1 && bxstride == 1)
{
@@ -1699,11 +1748,11 @@ matmul_c4_avx512f (gfc_array_c4 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_4) 0;
for (n = 0; n < count; n++)
s += abase_x[n] * bbase_y[n];
@@ -1718,11 +1767,11 @@ matmul_c4_avx512f (gfc_array_c4 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_4) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n];
- dest[y*rystride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n) * bbase_y[n];
+ ARRAY_ELEM_AT_OFFSET (dest, y * rystride_bytes) = s;
}
}
}
@@ -1733,26 +1782,27 @@ matmul_c4_avx512f (gfc_array_c4 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_4) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n*bxstride];
- dest[y*rxstride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
}
}
else if (axstride < aystride)
{
for (y = 0; y < ycount; y++)
for (x = 0; x < xcount; x++)
- dest[x*rxstride + y*rystride] = (GFC_COMPLEX_4)0;
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y) = (GFC_COMPLEX_4)0;
for (y = 0; y < ycount; y++)
for (n = 0; n < count; n++)
for (x = 0; x < xcount; x++)
/* dest[x,y] += a[x,n] * b[n,y] */
- dest[x*rxstride + y*rystride] +=
- abase[x*axstride + n*aystride] *
- bbase[n*bxstride + y*bystride];
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y)
+ += GFC_DESCRIPTOR2_ELEM (a, x, n)
+ * GFC_DESCRIPTOR2_ELEM (b, n, y);
}
else
{
@@ -1763,15 +1813,16 @@ matmul_c4_avx512f (gfc_array_c4 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_4) 0;
for (n = 0; n < count; n++)
- s += abase_x[n*aystride] * bbase_y[n*bxstride];
- dest_y[x*rxstride] = s;
+ s += ARRAY_ELEM_AT_OFFSET (abase_x, n * aystride_bytes)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ ARRAY_ELEM_AT_OFFSET (dest_y, x * rxstride_bytes) = s;
}
}
}
@@ -1812,7 +1863,8 @@ matmul_c4_vanilla (gfc_array_c4 * const restrict retarray,
index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
index_type x, y, n, count, xcount, ycount;
- index_type aystride_bytes, bystride_bytes, rystride_bytes;
+ index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
+ rxstride_bytes, rystride_bytes;
assert (GFC_DESCRIPTOR_RANK (a) == 2
|| GFC_DESCRIPTOR_RANK (b) == 2);
@@ -1901,12 +1953,13 @@ matmul_c4_vanilla (gfc_array_c4 * const restrict retarray,
either as a row or a column matrix. We want both cases to
work. */
rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
+ rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
}
else
{
rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
+ rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
}
@@ -1915,6 +1968,7 @@ matmul_c4_vanilla (gfc_array_c4 * const restrict retarray,
/* Treat it as a a row matrix A[1,count]. */
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = 1;
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = sizeof (GFC_COMPLEX_4);
xcount = 1;
@@ -1924,6 +1978,7 @@ matmul_c4_vanilla (gfc_array_c4 * const restrict retarray,
{
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = GFC_DESCRIPTOR_STRIDE(a,1);
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,1);
count = GFC_DESCRIPTOR_EXTENT(a,1);
@@ -1942,17 +1997,20 @@ matmul_c4_vanilla (gfc_array_c4 * const restrict retarray,
{
/* Treat it as a column matrix B[count,1] */
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
/* bystride should never be used for 1-dimensional b.
The value is only used for calculation of the
memory by the buffer. */
bystride = 256;
+ bystride_bytes = 99999999;
ycount = 1;
}
else
{
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
bystride = GFC_DESCRIPTOR_STRIDE(b,1);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
bystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,1);
ycount = GFC_DESCRIPTOR_EXTENT(b,1);
}
@@ -2012,12 +2070,11 @@ matmul_c4_vanilla (gfc_array_c4 * const restrict retarray,
from netlib.org, translated to C, and modified for matmul.m4. */
- const GFC_COMPLEX_4 *a, *b;
GFC_COMPLEX_4 *c;
const index_type m = xcount, n = ycount, k = count;
/* System generated locals */
- index_type a_dim1, b_dim1, c_dim1,
+ index_type a_dim1, b_dim1,
i1, i2, i3, i4, i5, i6;
/* Local variables */
@@ -2027,19 +2084,25 @@ matmul_c4_vanilla (gfc_array_c4 * const restrict retarray,
index_type isec, jsec, lsec, uisec, ujsec, ulsec;
GFC_COMPLEX_4 *t1;
- a = abase;
- b = bbase;
c = retarray->base_addr;
/* Parameter adjustments */
- c_dim1 = rystride;
a_dim1 = aystride;
b_dim1 = bystride;
- /* Empty c first. */
+#define A_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (abase, (i) * sizeof (GFC_COMPLEX_4) + (j) * aystride_bytes))
+
+#define B_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (bbase, (i) * sizeof (GFC_COMPLEX_4) + (j) * bystride_bytes))
+
+#define C_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (c, (i) * sizeof (GFC_COMPLEX_4) + (j) * rystride_bytes))
+
+ /* Empty result first. */
for (j=0; j<n; j++)
for (i=0; i<m; i++)
- c[i + j * c_dim1] = (GFC_COMPLEX_4)0;
+ C_ARRAY_ELEM (i, j) = (GFC_COMPLEX_4)0;
/* Early exit if possible */
if (m == 0 || n == 0 || k == 0)
@@ -2091,20 +2154,20 @@ matmul_c4_vanilla (gfc_array_c4 * const restrict retarray,
for (i = ii; i < i5; i += 2)
{
t1[l - ll + 1 + ((i - ii + 1) << 8) - 257] =
- a[i + l * a_dim1];
+ A_ARRAY_ELEM (i, l);
t1[l - ll + 2 + ((i - ii + 1) << 8) - 257] =
- a[i + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i, l + 1);
t1[l - ll + 1 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + l * a_dim1];
+ A_ARRAY_ELEM (i + 1, l);
t1[l - ll + 2 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i + 1, l + 1);
}
if (uisec < isec)
{
t1[l - ll + 1 + (isec << 8) - 257] =
- a[ii + isec - 1 + l * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l);
t1[l - ll + 2 + (isec << 8) - 257] =
- a[ii + isec - 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l + 1);
}
}
if (ulsec < lsec)
@@ -2113,7 +2176,7 @@ matmul_c4_vanilla (gfc_array_c4 * const restrict retarray,
for (i = ii; i< i4; ++i)
{
t1[lsec + ((i - ii + 1) << 8) - 257] =
- a[i + (ll + lsec - 1) * a_dim1];
+ A_ARRAY_ELEM (i, ll + lsec - 1);
}
}
@@ -2124,100 +2187,100 @@ matmul_c4_vanilla (gfc_array_c4 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f22 = c[i + 1 + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f23 = c[i + 1 + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
- f24 = c[i + 1 + (j + 3) * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
- f32 = c[i + 2 + (j + 1) * c_dim1];
- f42 = c[i + 3 + (j + 1) * c_dim1];
- f33 = c[i + 2 + (j + 2) * c_dim1];
- f43 = c[i + 3 + (j + 2) * c_dim1];
- f34 = c[i + 2 + (j + 3) * c_dim1];
- f44 = c[i + 3 + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f22 = C_ARRAY_ELEM (i + 1, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f23 = C_ARRAY_ELEM (i + 1, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
+ f24 = C_ARRAY_ELEM (i + 1, j + 3);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
+ f32 = C_ARRAY_ELEM (i + 2, j + 1);
+ f42 = C_ARRAY_ELEM (i + 3, j + 1);
+ f33 = C_ARRAY_ELEM (i + 2, j + 2);
+ f43 = C_ARRAY_ELEM (i + 3, j + 2);
+ f34 = C_ARRAY_ELEM (i + 2, j + 3);
+ f44 = C_ARRAY_ELEM (i + 3, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f22 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f23 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f24 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f32 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f42 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f33 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f43 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f34 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f44 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + 1 + (j + 1) * c_dim1] = f22;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + 1 + (j + 2) * c_dim1] = f23;
- c[i + (j + 3) * c_dim1] = f14;
- c[i + 1 + (j + 3) * c_dim1] = f24;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
- c[i + 2 + (j + 1) * c_dim1] = f32;
- c[i + 3 + (j + 1) * c_dim1] = f42;
- c[i + 2 + (j + 2) * c_dim1] = f33;
- c[i + 3 + (j + 2) * c_dim1] = f43;
- c[i + 2 + (j + 3) * c_dim1] = f34;
- c[i + 3 + (j + 3) * c_dim1] = f44;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i + 1, j + 1) = f22;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i + 1, j + 2) = f23;
+ C_ARRAY_ELEM (i, j + 3) = f14;
+ C_ARRAY_ELEM (i + 1, j + 3) = f24;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
+ C_ARRAY_ELEM (i + 2, j + 1) = f32;
+ C_ARRAY_ELEM (i + 3, j + 1) = f42;
+ C_ARRAY_ELEM (i + 2, j + 2) = f33;
+ C_ARRAY_ELEM (i + 3, j + 2) = f43;
+ C_ARRAY_ELEM (i + 2, j + 3) = f34;
+ C_ARRAY_ELEM (i + 3, j + 3) = f44;
}
if (uisec < isec)
{
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 1) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 2) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 3) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + (j + 3) * c_dim1] = f14;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i, j + 3) = f14;
}
}
}
@@ -2229,38 +2292,38 @@ matmul_c4_vanilla (gfc_array_c4 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
}
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
+ C_ARRAY_ELEM (i, j) = f11;
}
}
}
@@ -2269,6 +2332,9 @@ matmul_c4_vanilla (gfc_array_c4 * const restrict retarray,
}
free(t1);
return;
+#undef A_ARRAY_ELEM
+#undef B_ARRAY_ELEM
+#undef C_ARRAY_ELEM
}
else if (rxstride == 1 && aystride == 1 && bxstride == 1)
{
@@ -2281,11 +2347,11 @@ matmul_c4_vanilla (gfc_array_c4 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_4) 0;
for (n = 0; n < count; n++)
s += abase_x[n] * bbase_y[n];
@@ -2300,11 +2366,11 @@ matmul_c4_vanilla (gfc_array_c4 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_4) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n];
- dest[y*rystride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n) * bbase_y[n];
+ ARRAY_ELEM_AT_OFFSET (dest, y * rystride_bytes) = s;
}
}
}
@@ -2315,26 +2381,27 @@ matmul_c4_vanilla (gfc_array_c4 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_4) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n*bxstride];
- dest[y*rxstride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
}
}
else if (axstride < aystride)
{
for (y = 0; y < ycount; y++)
for (x = 0; x < xcount; x++)
- dest[x*rxstride + y*rystride] = (GFC_COMPLEX_4)0;
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y) = (GFC_COMPLEX_4)0;
for (y = 0; y < ycount; y++)
for (n = 0; n < count; n++)
for (x = 0; x < xcount; x++)
/* dest[x,y] += a[x,n] * b[n,y] */
- dest[x*rxstride + y*rystride] +=
- abase[x*axstride + n*aystride] *
- bbase[n*bxstride + y*bystride];
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y)
+ += GFC_DESCRIPTOR2_ELEM (a, x, n)
+ * GFC_DESCRIPTOR2_ELEM (b, n, y);
}
else
{
@@ -2345,15 +2412,16 @@ matmul_c4_vanilla (gfc_array_c4 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_4) 0;
for (n = 0; n < count; n++)
- s += abase_x[n*aystride] * bbase_y[n*bxstride];
- dest_y[x*rxstride] = s;
+ s += ARRAY_ELEM_AT_OFFSET (abase_x, n * aystride_bytes)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ ARRAY_ELEM_AT_OFFSET (dest_y, x * rxstride_bytes) = s;
}
}
}
@@ -2453,7 +2521,8 @@ matmul_c4 (gfc_array_c4 * const restrict retarray,
index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
index_type x, y, n, count, xcount, ycount;
- index_type aystride_bytes, bystride_bytes, rystride_bytes;
+ index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
+ rxstride_bytes, rystride_bytes;
assert (GFC_DESCRIPTOR_RANK (a) == 2
|| GFC_DESCRIPTOR_RANK (b) == 2);
@@ -2542,12 +2611,13 @@ matmul_c4 (gfc_array_c4 * const restrict retarray,
either as a row or a column matrix. We want both cases to
work. */
rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
+ rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
}
else
{
rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
+ rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
}
@@ -2556,6 +2626,7 @@ matmul_c4 (gfc_array_c4 * const restrict retarray,
/* Treat it as a a row matrix A[1,count]. */
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = 1;
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = sizeof (GFC_COMPLEX_4);
xcount = 1;
@@ -2565,6 +2636,7 @@ matmul_c4 (gfc_array_c4 * const restrict retarray,
{
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = GFC_DESCRIPTOR_STRIDE(a,1);
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,1);
count = GFC_DESCRIPTOR_EXTENT(a,1);
@@ -2583,17 +2655,20 @@ matmul_c4 (gfc_array_c4 * const restrict retarray,
{
/* Treat it as a column matrix B[count,1] */
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
/* bystride should never be used for 1-dimensional b.
The value is only used for calculation of the
memory by the buffer. */
bystride = 256;
+ bystride_bytes = 99999999;
ycount = 1;
}
else
{
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
bystride = GFC_DESCRIPTOR_STRIDE(b,1);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
bystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,1);
ycount = GFC_DESCRIPTOR_EXTENT(b,1);
}
@@ -2653,12 +2728,11 @@ matmul_c4 (gfc_array_c4 * const restrict retarray,
from netlib.org, translated to C, and modified for matmul.m4. */
- const GFC_COMPLEX_4 *a, *b;
GFC_COMPLEX_4 *c;
const index_type m = xcount, n = ycount, k = count;
/* System generated locals */
- index_type a_dim1, b_dim1, c_dim1,
+ index_type a_dim1, b_dim1,
i1, i2, i3, i4, i5, i6;
/* Local variables */
@@ -2668,19 +2742,25 @@ matmul_c4 (gfc_array_c4 * const restrict retarray,
index_type isec, jsec, lsec, uisec, ujsec, ulsec;
GFC_COMPLEX_4 *t1;
- a = abase;
- b = bbase;
c = retarray->base_addr;
/* Parameter adjustments */
- c_dim1 = rystride;
a_dim1 = aystride;
b_dim1 = bystride;
- /* Empty c first. */
+#define A_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (abase, (i) * sizeof (GFC_COMPLEX_4) + (j) * aystride_bytes))
+
+#define B_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (bbase, (i) * sizeof (GFC_COMPLEX_4) + (j) * bystride_bytes))
+
+#define C_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (c, (i) * sizeof (GFC_COMPLEX_4) + (j) * rystride_bytes))
+
+ /* Empty result first. */
for (j=0; j<n; j++)
for (i=0; i<m; i++)
- c[i + j * c_dim1] = (GFC_COMPLEX_4)0;
+ C_ARRAY_ELEM (i, j) = (GFC_COMPLEX_4)0;
/* Early exit if possible */
if (m == 0 || n == 0 || k == 0)
@@ -2732,20 +2812,20 @@ matmul_c4 (gfc_array_c4 * const restrict retarray,
for (i = ii; i < i5; i += 2)
{
t1[l - ll + 1 + ((i - ii + 1) << 8) - 257] =
- a[i + l * a_dim1];
+ A_ARRAY_ELEM (i, l);
t1[l - ll + 2 + ((i - ii + 1) << 8) - 257] =
- a[i + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i, l + 1);
t1[l - ll + 1 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + l * a_dim1];
+ A_ARRAY_ELEM (i + 1, l);
t1[l - ll + 2 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i + 1, l + 1);
}
if (uisec < isec)
{
t1[l - ll + 1 + (isec << 8) - 257] =
- a[ii + isec - 1 + l * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l);
t1[l - ll + 2 + (isec << 8) - 257] =
- a[ii + isec - 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l + 1);
}
}
if (ulsec < lsec)
@@ -2754,7 +2834,7 @@ matmul_c4 (gfc_array_c4 * const restrict retarray,
for (i = ii; i< i4; ++i)
{
t1[lsec + ((i - ii + 1) << 8) - 257] =
- a[i + (ll + lsec - 1) * a_dim1];
+ A_ARRAY_ELEM (i, ll + lsec - 1);
}
}
@@ -2765,100 +2845,100 @@ matmul_c4 (gfc_array_c4 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f22 = c[i + 1 + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f23 = c[i + 1 + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
- f24 = c[i + 1 + (j + 3) * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
- f32 = c[i + 2 + (j + 1) * c_dim1];
- f42 = c[i + 3 + (j + 1) * c_dim1];
- f33 = c[i + 2 + (j + 2) * c_dim1];
- f43 = c[i + 3 + (j + 2) * c_dim1];
- f34 = c[i + 2 + (j + 3) * c_dim1];
- f44 = c[i + 3 + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f22 = C_ARRAY_ELEM (i + 1, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f23 = C_ARRAY_ELEM (i + 1, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
+ f24 = C_ARRAY_ELEM (i + 1, j + 3);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
+ f32 = C_ARRAY_ELEM (i + 2, j + 1);
+ f42 = C_ARRAY_ELEM (i + 3, j + 1);
+ f33 = C_ARRAY_ELEM (i + 2, j + 2);
+ f43 = C_ARRAY_ELEM (i + 3, j + 2);
+ f34 = C_ARRAY_ELEM (i + 2, j + 3);
+ f44 = C_ARRAY_ELEM (i + 3, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f22 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f23 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f24 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f32 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f42 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f33 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f43 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f34 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f44 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + 1 + (j + 1) * c_dim1] = f22;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + 1 + (j + 2) * c_dim1] = f23;
- c[i + (j + 3) * c_dim1] = f14;
- c[i + 1 + (j + 3) * c_dim1] = f24;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
- c[i + 2 + (j + 1) * c_dim1] = f32;
- c[i + 3 + (j + 1) * c_dim1] = f42;
- c[i + 2 + (j + 2) * c_dim1] = f33;
- c[i + 3 + (j + 2) * c_dim1] = f43;
- c[i + 2 + (j + 3) * c_dim1] = f34;
- c[i + 3 + (j + 3) * c_dim1] = f44;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i + 1, j + 1) = f22;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i + 1, j + 2) = f23;
+ C_ARRAY_ELEM (i, j + 3) = f14;
+ C_ARRAY_ELEM (i + 1, j + 3) = f24;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
+ C_ARRAY_ELEM (i + 2, j + 1) = f32;
+ C_ARRAY_ELEM (i + 3, j + 1) = f42;
+ C_ARRAY_ELEM (i + 2, j + 2) = f33;
+ C_ARRAY_ELEM (i + 3, j + 2) = f43;
+ C_ARRAY_ELEM (i + 2, j + 3) = f34;
+ C_ARRAY_ELEM (i + 3, j + 3) = f44;
}
if (uisec < isec)
{
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 1) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 2) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 3) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + (j + 3) * c_dim1] = f14;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i, j + 3) = f14;
}
}
}
@@ -2870,38 +2950,38 @@ matmul_c4 (gfc_array_c4 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
}
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
+ C_ARRAY_ELEM (i, j) = f11;
}
}
}
@@ -2910,6 +2990,9 @@ matmul_c4 (gfc_array_c4 * const restrict retarray,
}
free(t1);
return;
+#undef A_ARRAY_ELEM
+#undef B_ARRAY_ELEM
+#undef C_ARRAY_ELEM
}
else if (rxstride == 1 && aystride == 1 && bxstride == 1)
{
@@ -2922,11 +3005,11 @@ matmul_c4 (gfc_array_c4 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_4) 0;
for (n = 0; n < count; n++)
s += abase_x[n] * bbase_y[n];
@@ -2941,11 +3024,11 @@ matmul_c4 (gfc_array_c4 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_4) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n];
- dest[y*rystride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n) * bbase_y[n];
+ ARRAY_ELEM_AT_OFFSET (dest, y * rystride_bytes) = s;
}
}
}
@@ -2956,26 +3039,27 @@ matmul_c4 (gfc_array_c4 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_4) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n*bxstride];
- dest[y*rxstride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
}
}
else if (axstride < aystride)
{
for (y = 0; y < ycount; y++)
for (x = 0; x < xcount; x++)
- dest[x*rxstride + y*rystride] = (GFC_COMPLEX_4)0;
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y) = (GFC_COMPLEX_4)0;
for (y = 0; y < ycount; y++)
for (n = 0; n < count; n++)
for (x = 0; x < xcount; x++)
/* dest[x,y] += a[x,n] * b[n,y] */
- dest[x*rxstride + y*rystride] +=
- abase[x*axstride + n*aystride] *
- bbase[n*bxstride + y*bystride];
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y)
+ += GFC_DESCRIPTOR2_ELEM (a, x, n)
+ * GFC_DESCRIPTOR2_ELEM (b, n, y);
}
else
{
@@ -2986,15 +3070,16 @@ matmul_c4 (gfc_array_c4 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_4) 0;
for (n = 0; n < count; n++)
- s += abase_x[n*aystride] * bbase_y[n*bxstride];
- dest_y[x*rxstride] = s;
+ s += ARRAY_ELEM_AT_OFFSET (abase_x, n * aystride_bytes)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ ARRAY_ELEM_AT_OFFSET (dest_y, x * rxstride_bytes) = s;
}
}
}
diff --git a/libgfortran/generated/matmul_c8.c b/libgfortran/generated/matmul_c8.c
index 66e14224cf1d..bc37144bfbbd 100644
--- a/libgfortran/generated/matmul_c8.c
+++ b/libgfortran/generated/matmul_c8.c
@@ -94,7 +94,8 @@ matmul_c8_avx (gfc_array_c8 * const restrict retarray,
index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
index_type x, y, n, count, xcount, ycount;
- index_type aystride_bytes, bystride_bytes, rystride_bytes;
+ index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
+ rxstride_bytes, rystride_bytes;
assert (GFC_DESCRIPTOR_RANK (a) == 2
|| GFC_DESCRIPTOR_RANK (b) == 2);
@@ -183,12 +184,13 @@ matmul_c8_avx (gfc_array_c8 * const restrict retarray,
either as a row or a column matrix. We want both cases to
work. */
rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
+ rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
}
else
{
rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
+ rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
}
@@ -197,6 +199,7 @@ matmul_c8_avx (gfc_array_c8 * const restrict retarray,
/* Treat it as a a row matrix A[1,count]. */
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = 1;
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = sizeof (GFC_COMPLEX_8);
xcount = 1;
@@ -206,6 +209,7 @@ matmul_c8_avx (gfc_array_c8 * const restrict retarray,
{
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = GFC_DESCRIPTOR_STRIDE(a,1);
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,1);
count = GFC_DESCRIPTOR_EXTENT(a,1);
@@ -224,17 +228,20 @@ matmul_c8_avx (gfc_array_c8 * const restrict retarray,
{
/* Treat it as a column matrix B[count,1] */
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
/* bystride should never be used for 1-dimensional b.
The value is only used for calculation of the
memory by the buffer. */
bystride = 256;
+ bystride_bytes = 99999999;
ycount = 1;
}
else
{
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
bystride = GFC_DESCRIPTOR_STRIDE(b,1);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
bystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,1);
ycount = GFC_DESCRIPTOR_EXTENT(b,1);
}
@@ -294,12 +301,11 @@ matmul_c8_avx (gfc_array_c8 * const restrict retarray,
from netlib.org, translated to C, and modified for matmul.m4. */
- const GFC_COMPLEX_8 *a, *b;
GFC_COMPLEX_8 *c;
const index_type m = xcount, n = ycount, k = count;
/* System generated locals */
- index_type a_dim1, b_dim1, c_dim1,
+ index_type a_dim1, b_dim1,
i1, i2, i3, i4, i5, i6;
/* Local variables */
@@ -309,19 +315,25 @@ matmul_c8_avx (gfc_array_c8 * const restrict retarray,
index_type isec, jsec, lsec, uisec, ujsec, ulsec;
GFC_COMPLEX_8 *t1;
- a = abase;
- b = bbase;
c = retarray->base_addr;
/* Parameter adjustments */
- c_dim1 = rystride;
a_dim1 = aystride;
b_dim1 = bystride;
- /* Empty c first. */
+#define A_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (abase, (i) * sizeof (GFC_COMPLEX_8) + (j) * aystride_bytes))
+
+#define B_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (bbase, (i) * sizeof (GFC_COMPLEX_8) + (j) * bystride_bytes))
+
+#define C_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (c, (i) * sizeof (GFC_COMPLEX_8) + (j) * rystride_bytes))
+
+ /* Empty result first. */
for (j=0; j<n; j++)
for (i=0; i<m; i++)
- c[i + j * c_dim1] = (GFC_COMPLEX_8)0;
+ C_ARRAY_ELEM (i, j) = (GFC_COMPLEX_8)0;
/* Early exit if possible */
if (m == 0 || n == 0 || k == 0)
@@ -373,20 +385,20 @@ matmul_c8_avx (gfc_array_c8 * const restrict retarray,
for (i = ii; i < i5; i += 2)
{
t1[l - ll + 1 + ((i - ii + 1) << 8) - 257] =
- a[i + l * a_dim1];
+ A_ARRAY_ELEM (i, l);
t1[l - ll + 2 + ((i - ii + 1) << 8) - 257] =
- a[i + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i, l + 1);
t1[l - ll + 1 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + l * a_dim1];
+ A_ARRAY_ELEM (i + 1, l);
t1[l - ll + 2 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i + 1, l + 1);
}
if (uisec < isec)
{
t1[l - ll + 1 + (isec << 8) - 257] =
- a[ii + isec - 1 + l * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l);
t1[l - ll + 2 + (isec << 8) - 257] =
- a[ii + isec - 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l + 1);
}
}
if (ulsec < lsec)
@@ -395,7 +407,7 @@ matmul_c8_avx (gfc_array_c8 * const restrict retarray,
for (i = ii; i< i4; ++i)
{
t1[lsec + ((i - ii + 1) << 8) - 257] =
- a[i + (ll + lsec - 1) * a_dim1];
+ A_ARRAY_ELEM (i, ll + lsec - 1);
}
}
@@ -406,100 +418,100 @@ matmul_c8_avx (gfc_array_c8 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f22 = c[i + 1 + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f23 = c[i + 1 + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
- f24 = c[i + 1 + (j + 3) * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
- f32 = c[i + 2 + (j + 1) * c_dim1];
- f42 = c[i + 3 + (j + 1) * c_dim1];
- f33 = c[i + 2 + (j + 2) * c_dim1];
- f43 = c[i + 3 + (j + 2) * c_dim1];
- f34 = c[i + 2 + (j + 3) * c_dim1];
- f44 = c[i + 3 + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f22 = C_ARRAY_ELEM (i + 1, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f23 = C_ARRAY_ELEM (i + 1, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
+ f24 = C_ARRAY_ELEM (i + 1, j + 3);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
+ f32 = C_ARRAY_ELEM (i + 2, j + 1);
+ f42 = C_ARRAY_ELEM (i + 3, j + 1);
+ f33 = C_ARRAY_ELEM (i + 2, j + 2);
+ f43 = C_ARRAY_ELEM (i + 3, j + 2);
+ f34 = C_ARRAY_ELEM (i + 2, j + 3);
+ f44 = C_ARRAY_ELEM (i + 3, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f22 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f23 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f24 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f32 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f42 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f33 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f43 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f34 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f44 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + 1 + (j + 1) * c_dim1] = f22;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + 1 + (j + 2) * c_dim1] = f23;
- c[i + (j + 3) * c_dim1] = f14;
- c[i + 1 + (j + 3) * c_dim1] = f24;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
- c[i + 2 + (j + 1) * c_dim1] = f32;
- c[i + 3 + (j + 1) * c_dim1] = f42;
- c[i + 2 + (j + 2) * c_dim1] = f33;
- c[i + 3 + (j + 2) * c_dim1] = f43;
- c[i + 2 + (j + 3) * c_dim1] = f34;
- c[i + 3 + (j + 3) * c_dim1] = f44;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i + 1, j + 1) = f22;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i + 1, j + 2) = f23;
+ C_ARRAY_ELEM (i, j + 3) = f14;
+ C_ARRAY_ELEM (i + 1, j + 3) = f24;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
+ C_ARRAY_ELEM (i + 2, j + 1) = f32;
+ C_ARRAY_ELEM (i + 3, j + 1) = f42;
+ C_ARRAY_ELEM (i + 2, j + 2) = f33;
+ C_ARRAY_ELEM (i + 3, j + 2) = f43;
+ C_ARRAY_ELEM (i + 2, j + 3) = f34;
+ C_ARRAY_ELEM (i + 3, j + 3) = f44;
}
if (uisec < isec)
{
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 1) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 2) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 3) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + (j + 3) * c_dim1] = f14;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i, j + 3) = f14;
}
}
}
@@ -511,38 +523,38 @@ matmul_c8_avx (gfc_array_c8 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
}
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
+ C_ARRAY_ELEM (i, j) = f11;
}
}
}
@@ -551,6 +563,9 @@ matmul_c8_avx (gfc_array_c8 * const restrict retarray,
}
free(t1);
return;
+#undef A_ARRAY_ELEM
+#undef B_ARRAY_ELEM
+#undef C_ARRAY_ELEM
}
else if (rxstride == 1 && aystride == 1 && bxstride == 1)
{
@@ -563,11 +578,11 @@ matmul_c8_avx (gfc_array_c8 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_8) 0;
for (n = 0; n < count; n++)
s += abase_x[n] * bbase_y[n];
@@ -582,11 +597,11 @@ matmul_c8_avx (gfc_array_c8 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_8) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n];
- dest[y*rystride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n) * bbase_y[n];
+ ARRAY_ELEM_AT_OFFSET (dest, y * rystride_bytes) = s;
}
}
}
@@ -597,26 +612,27 @@ matmul_c8_avx (gfc_array_c8 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_8) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n*bxstride];
- dest[y*rxstride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
}
}
else if (axstride < aystride)
{
for (y = 0; y < ycount; y++)
for (x = 0; x < xcount; x++)
- dest[x*rxstride + y*rystride] = (GFC_COMPLEX_8)0;
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y) = (GFC_COMPLEX_8)0;
for (y = 0; y < ycount; y++)
for (n = 0; n < count; n++)
for (x = 0; x < xcount; x++)
/* dest[x,y] += a[x,n] * b[n,y] */
- dest[x*rxstride + y*rystride] +=
- abase[x*axstride + n*aystride] *
- bbase[n*bxstride + y*bystride];
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y)
+ += GFC_DESCRIPTOR2_ELEM (a, x, n)
+ * GFC_DESCRIPTOR2_ELEM (b, n, y);
}
else
{
@@ -627,15 +643,16 @@ matmul_c8_avx (gfc_array_c8 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_8) 0;
for (n = 0; n < count; n++)
- s += abase_x[n*aystride] * bbase_y[n*bxstride];
- dest_y[x*rxstride] = s;
+ s += ARRAY_ELEM_AT_OFFSET (abase_x, n * aystride_bytes)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ ARRAY_ELEM_AT_OFFSET (dest_y, x * rxstride_bytes) = s;
}
}
}
@@ -662,7 +679,8 @@ matmul_c8_avx2 (gfc_array_c8 * const restrict retarray,
index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
index_type x, y, n, count, xcount, ycount;
- index_type aystride_bytes, bystride_bytes, rystride_bytes;
+ index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
+ rxstride_bytes, rystride_bytes;
assert (GFC_DESCRIPTOR_RANK (a) == 2
|| GFC_DESCRIPTOR_RANK (b) == 2);
@@ -751,12 +769,13 @@ matmul_c8_avx2 (gfc_array_c8 * const restrict retarray,
either as a row or a column matrix. We want both cases to
work. */
rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
+ rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
}
else
{
rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
+ rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
}
@@ -765,6 +784,7 @@ matmul_c8_avx2 (gfc_array_c8 * const restrict retarray,
/* Treat it as a a row matrix A[1,count]. */
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = 1;
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = sizeof (GFC_COMPLEX_8);
xcount = 1;
@@ -774,6 +794,7 @@ matmul_c8_avx2 (gfc_array_c8 * const restrict retarray,
{
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = GFC_DESCRIPTOR_STRIDE(a,1);
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,1);
count = GFC_DESCRIPTOR_EXTENT(a,1);
@@ -792,17 +813,20 @@ matmul_c8_avx2 (gfc_array_c8 * const restrict retarray,
{
/* Treat it as a column matrix B[count,1] */
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
/* bystride should never be used for 1-dimensional b.
The value is only used for calculation of the
memory by the buffer. */
bystride = 256;
+ bystride_bytes = 99999999;
ycount = 1;
}
else
{
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
bystride = GFC_DESCRIPTOR_STRIDE(b,1);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
bystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,1);
ycount = GFC_DESCRIPTOR_EXTENT(b,1);
}
@@ -862,12 +886,11 @@ matmul_c8_avx2 (gfc_array_c8 * const restrict retarray,
from netlib.org, translated to C, and modified for matmul.m4. */
- const GFC_COMPLEX_8 *a, *b;
GFC_COMPLEX_8 *c;
const index_type m = xcount, n = ycount, k = count;
/* System generated locals */
- index_type a_dim1, b_dim1, c_dim1,
+ index_type a_dim1, b_dim1,
i1, i2, i3, i4, i5, i6;
/* Local variables */
@@ -877,19 +900,25 @@ matmul_c8_avx2 (gfc_array_c8 * const restrict retarray,
index_type isec, jsec, lsec, uisec, ujsec, ulsec;
GFC_COMPLEX_8 *t1;
- a = abase;
- b = bbase;
c = retarray->base_addr;
/* Parameter adjustments */
- c_dim1 = rystride;
a_dim1 = aystride;
b_dim1 = bystride;
- /* Empty c first. */
+#define A_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (abase, (i) * sizeof (GFC_COMPLEX_8) + (j) * aystride_bytes))
+
+#define B_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (bbase, (i) * sizeof (GFC_COMPLEX_8) + (j) * bystride_bytes))
+
+#define C_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (c, (i) * sizeof (GFC_COMPLEX_8) + (j) * rystride_bytes))
+
+ /* Empty result first. */
for (j=0; j<n; j++)
for (i=0; i<m; i++)
- c[i + j * c_dim1] = (GFC_COMPLEX_8)0;
+ C_ARRAY_ELEM (i, j) = (GFC_COMPLEX_8)0;
/* Early exit if possible */
if (m == 0 || n == 0 || k == 0)
@@ -941,20 +970,20 @@ matmul_c8_avx2 (gfc_array_c8 * const restrict retarray,
for (i = ii; i < i5; i += 2)
{
t1[l - ll + 1 + ((i - ii + 1) << 8) - 257] =
- a[i + l * a_dim1];
+ A_ARRAY_ELEM (i, l);
t1[l - ll + 2 + ((i - ii + 1) << 8) - 257] =
- a[i + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i, l + 1);
t1[l - ll + 1 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + l * a_dim1];
+ A_ARRAY_ELEM (i + 1, l);
t1[l - ll + 2 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i + 1, l + 1);
}
if (uisec < isec)
{
t1[l - ll + 1 + (isec << 8) - 257] =
- a[ii + isec - 1 + l * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l);
t1[l - ll + 2 + (isec << 8) - 257] =
- a[ii + isec - 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l + 1);
}
}
if (ulsec < lsec)
@@ -963,7 +992,7 @@ matmul_c8_avx2 (gfc_array_c8 * const restrict retarray,
for (i = ii; i< i4; ++i)
{
t1[lsec + ((i - ii + 1) << 8) - 257] =
- a[i + (ll + lsec - 1) * a_dim1];
+ A_ARRAY_ELEM (i, ll + lsec - 1);
}
}
@@ -974,100 +1003,100 @@ matmul_c8_avx2 (gfc_array_c8 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f22 = c[i + 1 + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f23 = c[i + 1 + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
- f24 = c[i + 1 + (j + 3) * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
- f32 = c[i + 2 + (j + 1) * c_dim1];
- f42 = c[i + 3 + (j + 1) * c_dim1];
- f33 = c[i + 2 + (j + 2) * c_dim1];
- f43 = c[i + 3 + (j + 2) * c_dim1];
- f34 = c[i + 2 + (j + 3) * c_dim1];
- f44 = c[i + 3 + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f22 = C_ARRAY_ELEM (i + 1, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f23 = C_ARRAY_ELEM (i + 1, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
+ f24 = C_ARRAY_ELEM (i + 1, j + 3);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
+ f32 = C_ARRAY_ELEM (i + 2, j + 1);
+ f42 = C_ARRAY_ELEM (i + 3, j + 1);
+ f33 = C_ARRAY_ELEM (i + 2, j + 2);
+ f43 = C_ARRAY_ELEM (i + 3, j + 2);
+ f34 = C_ARRAY_ELEM (i + 2, j + 3);
+ f44 = C_ARRAY_ELEM (i + 3, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f22 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f23 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f24 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f32 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f42 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f33 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f43 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f34 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f44 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + 1 + (j + 1) * c_dim1] = f22;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + 1 + (j + 2) * c_dim1] = f23;
- c[i + (j + 3) * c_dim1] = f14;
- c[i + 1 + (j + 3) * c_dim1] = f24;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
- c[i + 2 + (j + 1) * c_dim1] = f32;
- c[i + 3 + (j + 1) * c_dim1] = f42;
- c[i + 2 + (j + 2) * c_dim1] = f33;
- c[i + 3 + (j + 2) * c_dim1] = f43;
- c[i + 2 + (j + 3) * c_dim1] = f34;
- c[i + 3 + (j + 3) * c_dim1] = f44;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i + 1, j + 1) = f22;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i + 1, j + 2) = f23;
+ C_ARRAY_ELEM (i, j + 3) = f14;
+ C_ARRAY_ELEM (i + 1, j + 3) = f24;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
+ C_ARRAY_ELEM (i + 2, j + 1) = f32;
+ C_ARRAY_ELEM (i + 3, j + 1) = f42;
+ C_ARRAY_ELEM (i + 2, j + 2) = f33;
+ C_ARRAY_ELEM (i + 3, j + 2) = f43;
+ C_ARRAY_ELEM (i + 2, j + 3) = f34;
+ C_ARRAY_ELEM (i + 3, j + 3) = f44;
}
if (uisec < isec)
{
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 1) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 2) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + (j + 3) * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + (j + 3) * c_dim1] = f14;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i, j + 3) = f14;
}
}
}
@@ -1079,38 +1108,38 @@ matmul_c8_avx2 (gfc_array_c8 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
}
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- 257] * b[l + j * b_dim1];
+ 257] * B_ARRAY_ELEM (l, j);
}
- c[i + j * c_dim1] = f11;
+ C_ARRAY_ELEM (i, j) = f11;
}
}
}
@@ -1119,6 +1148,9 @@ matmul_c8_avx2 (gfc_array_c8 * const restrict retarray,
}
free(t1);
return;
+#undef A_ARRAY_ELEM
+#undef B_ARRAY_ELEM
+#undef C_ARRAY_ELEM
}
else if (rxstride == 1 && aystride == 1 && bxstride == 1)
{
@@ -1131,11 +1163,11 @@ matmul_c8_avx2 (gfc_array_c8 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_8) 0;
for (n = 0; n < count; n++)
s += abase_x[n] * bbase_y[n];
@@ -1150,11 +1182,11 @@ matmul_c8_avx2 (gfc_array_c8 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_8) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n];
- dest[y*rystride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n) * bbase_y[n];
+ ARRAY_ELEM_AT_OFFSET (dest, y * rystride_bytes) = s;
}
}
}
@@ -1165,26 +1197,27 @@ matmul_c8_avx2 (gfc_array_c8 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
s = (GFC_COMPLEX_8) 0;
for (n = 0; n < count; n++)
- s += abase[n*axstride] * bbase_y[n*bxstride];
- dest[y*rxstride] = s;
+ s += GFC_DESCRIPTOR1_ELEM (a, n)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
}
}
else if (axstride < aystride)
{
for (y = 0; y < ycount; y++)
for (x = 0; x < xcount; x++)
- dest[x*rxstride + y*rystride] = (GFC_COMPLEX_8)0;
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y) = (GFC_COMPLEX_8)0;
for (y = 0; y < ycount; y++)
for (n = 0; n < count; n++)
for (x = 0; x < xcount; x++)
/* dest[x,y] += a[x,n] * b[n,y] */
- dest[x*rxstride + y*rystride] +=
- abase[x*axstride + n*aystride] *
- bbase[n*bxstride + y*bystride];
+ GFC_DESCRIPTOR2_ELEM (retarray, x, y)
+ += GFC_DESCRIPTOR2_ELEM (a, x, n)
+ * GFC_DESCRIPTOR2_ELEM (b, n, y);
}
else
{
@@ -1195,15 +1228,16 @@ matmul_c8_avx2 (gfc_array_c8 * const restrict retarray,
for (y = 0; y < ycount; y++)
{
- bbase_y = &bbase[y*bystride];
- dest_y = &dest[y*rystride];
+ bbase_y = PTR_ADD_OFFSET (bbase, y * bystride_bytes);
+ dest_y = PTR_ADD_OFFSET (dest, y * rystride_bytes);
for (x = 0; x < xcount; x++)
{
- abase_x = &abase[x*axstride];
+ abase_x = PTR_ADD_OFFSET (abase, x * axstride_bytes);
s = (GFC_COMPLEX_8) 0;
for (n = 0; n < count; n++)
- s += abase_x[n*aystride] * bbase_y[n*bxstride];
- dest_y[x*rxstride] = s;
+ s += ARRAY_ELEM_AT_OFFSET (abase_x, n * aystride_bytes)
+ * ARRAY_ELEM_AT_OFFSET (bbase_y, n * bxstride_bytes);
+ ARRAY_ELEM_AT_OFFSET (dest_y, x * rxstride_bytes) = s;
}
}
}
@@ -1230,7 +1264,8 @@ matmul_c8_avx512f (gfc_array_c8 * const restrict retarray,
index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
index_type x, y, n, count, xcount, ycount;
- index_type aystride_bytes, bystride_bytes, rystride_bytes;
+ index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
+ rxstride_bytes, rystride_bytes;
assert (GFC_DESCRIPTOR_RANK (a) == 2
|| GFC_DESCRIPTOR_RANK (b) == 2);
@@ -1319,12 +1354,13 @@ matmul_c8_avx512f (gfc_array_c8 * const restrict retarray,
either as a row or a column matrix. We want both cases to
work. */
rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
- rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
+ rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
}
else
{
rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
+ rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
}
@@ -1333,6 +1369,7 @@ matmul_c8_avx512f (gfc_array_c8 * const restrict retarray,
/* Treat it as a a row matrix A[1,count]. */
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = 1;
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = sizeof (GFC_COMPLEX_8);
xcount = 1;
@@ -1342,6 +1379,7 @@ matmul_c8_avx512f (gfc_array_c8 * const restrict retarray,
{
axstride = GFC_DESCRIPTOR_STRIDE(a,0);
aystride = GFC_DESCRIPTOR_STRIDE(a,1);
+ axstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,0);
aystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(a,1);
count = GFC_DESCRIPTOR_EXTENT(a,1);
@@ -1360,17 +1398,20 @@ matmul_c8_avx512f (gfc_array_c8 * const restrict retarray,
{
/* Treat it as a column matrix B[count,1] */
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
/* bystride should never be used for 1-dimensional b.
The value is only used for calculation of the
memory by the buffer. */
bystride = 256;
+ bystride_bytes = 99999999;
ycount = 1;
}
else
{
bxstride = GFC_DESCRIPTOR_STRIDE(b,0);
bystride = GFC_DESCRIPTOR_STRIDE(b,1);
+ bxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,0);
bystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(b,1);
ycount = GFC_DESCRIPTOR_EXTENT(b,1);
}
@@ -1430,12 +1471,11 @@ matmul_c8_avx512f (gfc_array_c8 * const restrict retarray,
from netlib.org, translated to C, and modified for matmul.m4. */
- const GFC_COMPLEX_8 *a, *b;
GFC_COMPLEX_8 *c;
const index_type m = xcount, n = ycount, k = count;
/* System generated locals */
- index_type a_dim1, b_dim1, c_dim1,
+ index_type a_dim1, b_dim1,
i1, i2, i3, i4, i5, i6;
/* Local variables */
@@ -1445,19 +1485,25 @@ matmul_c8_avx512f (gfc_array_c8 * const restrict retarray,
index_type isec, jsec, lsec, uisec, ujsec, ulsec;
GFC_COMPLEX_8 *t1;
- a = abase;
- b = bbase;
c = retarray->base_addr;
/* Parameter adjustments */
- c_dim1 = rystride;
a_dim1 = aystride;
b_dim1 = bystride;
- /* Empty c first. */
+#define A_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (abase, (i) * sizeof (GFC_COMPLEX_8) + (j) * aystride_bytes))
+
+#define B_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (bbase, (i) * sizeof (GFC_COMPLEX_8) + (j) * bystride_bytes))
+
+#define C_ARRAY_ELEM(i,j) \
+ (ARRAY_ELEM_AT_OFFSET (c, (i) * sizeof (GFC_COMPLEX_8) + (j) * rystride_bytes))
+
+ /* Empty result first. */
for (j=0; j<n; j++)
for (i=0; i<m; i++)
- c[i + j * c_dim1] = (GFC_COMPLEX_8)0;
+ C_ARRAY_ELEM (i, j) = (GFC_COMPLEX_8)0;
/* Early exit if possible */
if (m == 0 || n == 0 || k == 0)
@@ -1509,20 +1555,20 @@ matmul_c8_avx512f (gfc_array_c8 * const restrict retarray,
for (i = ii; i < i5; i += 2)
{
t1[l - ll + 1 + ((i - ii + 1) << 8) - 257] =
- a[i + l * a_dim1];
+ A_ARRAY_ELEM (i, l);
t1[l - ll + 2 + ((i - ii + 1) << 8) - 257] =
- a[i + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i, l + 1);
t1[l - ll + 1 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + l * a_dim1];
+ A_ARRAY_ELEM (i + 1, l);
t1[l - ll + 2 + ((i - ii + 2) << 8) - 257] =
- a[i + 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (i + 1, l + 1);
}
if (uisec < isec)
{
t1[l - ll + 1 + (isec << 8) - 257] =
- a[ii + isec - 1 + l * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l);
t1[l - ll + 2 + (isec << 8) - 257] =
- a[ii + isec - 1 + (l + 1) * a_dim1];
+ A_ARRAY_ELEM (ii + isec - 1, l + 1);
}
}
if (ulsec < lsec)
@@ -1531,7 +1577,7 @@ matmul_c8_avx512f (gfc_array_c8 * const restrict retarray,
for (i = ii; i< i4; ++i)
{
t1[lsec + ((i - ii + 1) << 8) - 257] =
- a[i + (ll + lsec - 1) * a_dim1];
+ A_ARRAY_ELEM (i, ll + lsec - 1);
}
}
@@ -1542,100 +1588,100 @@ matmul_c8_avx512f (gfc_array_c8 * const restrict retarray,
i5 = ii + uisec;
for (i = ii; i < i5; i += 4)
{
- f11 = c[i + j * c_dim1];
- f21 = c[i + 1 + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f22 = c[i + 1 + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f23 = c[i + 1 + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
- f24 = c[i + 1 + (j + 3) * c_dim1];
- f31 = c[i + 2 + j * c_dim1];
- f41 = c[i + 3 + j * c_dim1];
- f32 = c[i + 2 + (j + 1) * c_dim1];
- f42 = c[i + 3 + (j + 1) * c_dim1];
- f33 = c[i + 2 + (j + 2) * c_dim1];
- f43 = c[i + 3 + (j + 2) * c_dim1];
- f34 = c[i + 2 + (j + 3) * c_dim1];
- f44 = c[i + 3 + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f21 = C_ARRAY_ELEM (i + 1, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f22 = C_ARRAY_ELEM (i + 1, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f23 = C_ARRAY_ELEM (i + 1, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
+ f24 = C_ARRAY_ELEM (i + 1, j + 3);
+ f31 = C_ARRAY_ELEM (i + 2, j);
+ f41 = C_ARRAY_ELEM (i + 3, j);
+ f32 = C_ARRAY_ELEM (i + 2, j + 1);
+ f42 = C_ARRAY_ELEM (i + 3, j + 1);
+ f33 = C_ARRAY_ELEM (i + 2, j + 2);
+ f43 = C_ARRAY_ELEM (i + 3, j + 2);
+ f34 = C_ARRAY_ELEM (i + 2, j + 3);
+ f44 = C_ARRAY_ELEM (i + 3, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f21 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f12 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f22 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f13 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f23 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f14 += t1[l - ll + 1 + ((i - ii + 1) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f24 += t1[l - ll + 1 + ((i - ii + 2) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f31 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f41 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + j * b_dim1];
+ * B_ARRAY_ELEM (l, j);
f32 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f42 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 1) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 1);
f33 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f43 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 2) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 2);
f34 += t1[l - ll + 1 + ((i - ii + 3) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
f44 += t1[l - ll + 1 + ((i - ii + 4) << 8) - 257]
- * b[l + (j + 3) * b_dim1];
+ * B_ARRAY_ELEM (l, j + 3);
}
- c[i + j * c_dim1] = f11;
- c[i + 1 + j * c_dim1] = f21;
- c[i + (j + 1) * c_dim1] = f12;
- c[i + 1 + (j + 1) * c_dim1] = f22;
- c[i + (j + 2) * c_dim1] = f13;
- c[i + 1 + (j + 2) * c_dim1] = f23;
- c[i + (j + 3) * c_dim1] = f14;
- c[i + 1 + (j + 3) * c_dim1] = f24;
- c[i + 2 + j * c_dim1] = f31;
- c[i + 3 + j * c_dim1] = f41;
- c[i + 2 + (j + 1) * c_dim1] = f32;
- c[i + 3 + (j + 1) * c_dim1] = f42;
- c[i + 2 + (j + 2) * c_dim1] = f33;
- c[i + 3 + (j + 2) * c_dim1] = f43;
- c[i + 2 + (j + 3) * c_dim1] = f34;
- c[i + 3 + (j + 3) * c_dim1] = f44;
+ C_ARRAY_ELEM (i, j) = f11;
+ C_ARRAY_ELEM (i + 1, j) = f21;
+ C_ARRAY_ELEM (i, j + 1) = f12;
+ C_ARRAY_ELEM (i + 1, j + 1) = f22;
+ C_ARRAY_ELEM (i, j + 2) = f13;
+ C_ARRAY_ELEM (i + 1, j + 2) = f23;
+ C_ARRAY_ELEM (i, j + 3) = f14;
+ C_ARRAY_ELEM (i + 1, j + 3) = f24;
+ C_ARRAY_ELEM (i + 2, j) = f31;
+ C_ARRAY_ELEM (i + 3, j) = f41;
+ C_ARRAY_ELEM (i + 2, j + 1) = f32;
+ C_ARRAY_ELEM (i + 3, j + 1) = f42;
+ C_ARRAY_ELEM (i + 2, j + 2) = f33;
+ C_ARRAY_ELEM (i + 3, j + 2) = f43;
+ C_ARRAY_ELEM (i + 2, j + 3) = f34;
+ C_ARRAY_ELEM (i + 3, j + 3) = f44;
}
if (uisec < isec)
{
i5 = ii + isec;
for (i = ii + uisec; i < i5; ++i)
{
- f11 = c[i + j * c_dim1];
- f12 = c[i + (j + 1) * c_dim1];
- f13 = c[i + (j + 2) * c_dim1];
- f14 = c[i + (j + 3) * c_dim1];
+ f11 = C_ARRAY_ELEM (i, j);
+ f12 = C_ARRAY_ELEM (i, j + 1);
+ f13 = C_ARRAY_ELEM (i, j + 2);
+ f14 = C_ARRAY_ELEM (i, j + 3);
i6 = ll + lsec;
for (l = ll; l < i6; ++l)
{
f11 += t1[l - ll + 1 + ((i - ii + 1) << 8) -
- [...]
[diff truncated at 524288 bytes]
More information about the Gcc-cvs
mailing list