[gcc(refs/users/mikael/heads/refactor_descriptor_v08)] Régénération fichiers générés

Mikael Morin mikael@gcc.gnu.org
Sun Sep 14 16:38:15 GMT 2025


https://gcc.gnu.org/g:318608013ed975a98b33673112582881e5b49308

commit 318608013ed975a98b33673112582881e5b49308
Author: Mikael Morin <mikael@gcc.gnu.org>
Date:   Sat Sep 13 16:46:31 2025 +0200

    Régénération fichiers générés

Diff:
---
 libgfortran/generated/all_l1.c           |  16 ++-
 libgfortran/generated/all_l16.c          |  16 ++-
 libgfortran/generated/all_l2.c           |  16 ++-
 libgfortran/generated/all_l4.c           |  16 ++-
 libgfortran/generated/all_l8.c           |  16 ++-
 libgfortran/generated/any_l1.c           |  16 ++-
 libgfortran/generated/any_l16.c          |  16 ++-
 libgfortran/generated/any_l2.c           |  16 ++-
 libgfortran/generated/any_l4.c           |  16 ++-
 libgfortran/generated/any_l8.c           |  16 ++-
 libgfortran/generated/count_16_l.c       |  16 ++-
 libgfortran/generated/count_1_l.c        |  16 ++-
 libgfortran/generated/count_2_l.c        |  16 ++-
 libgfortran/generated/count_4_l.c        |  16 ++-
 libgfortran/generated/count_8_l.c        |  16 ++-
 libgfortran/generated/cshift0_c10.c      |  19 ++--
 libgfortran/generated/cshift0_c16.c      |  19 ++--
 libgfortran/generated/cshift0_c17.c      |  19 ++--
 libgfortran/generated/cshift0_c4.c       |  19 ++--
 libgfortran/generated/cshift0_c8.c       |  19 ++--
 libgfortran/generated/cshift0_i1.c       |  19 ++--
 libgfortran/generated/cshift0_i16.c      |  19 ++--
 libgfortran/generated/cshift0_i2.c       |  19 ++--
 libgfortran/generated/cshift0_i4.c       |  19 ++--
 libgfortran/generated/cshift0_i8.c       |  19 ++--
 libgfortran/generated/cshift0_r10.c      |  19 ++--
 libgfortran/generated/cshift0_r16.c      |  19 ++--
 libgfortran/generated/cshift0_r17.c      |  19 ++--
 libgfortran/generated/cshift0_r4.c       |  19 ++--
 libgfortran/generated/cshift0_r8.c       |  19 ++--
 libgfortran/generated/cshift1_16.c       |  12 +--
 libgfortran/generated/cshift1_4.c        |  12 +--
 libgfortran/generated/cshift1_8.c        |  12 +--
 libgfortran/generated/eoshift1_16.c      |  13 ++-
 libgfortran/generated/eoshift1_4.c       |  13 ++-
 libgfortran/generated/eoshift1_8.c       |  13 ++-
 libgfortran/generated/eoshift3_16.c      |  13 +--
 libgfortran/generated/eoshift3_4.c       |  13 +--
 libgfortran/generated/eoshift3_8.c       |  13 +--
 libgfortran/generated/findloc1_c10.c     |  48 ++++-----
 libgfortran/generated/findloc1_c16.c     |  48 ++++-----
 libgfortran/generated/findloc1_c17.c     |  48 ++++-----
 libgfortran/generated/findloc1_c4.c      |  48 ++++-----
 libgfortran/generated/findloc1_c8.c      |  48 ++++-----
 libgfortran/generated/findloc1_i1.c      |  48 ++++-----
 libgfortran/generated/findloc1_i16.c     |  48 ++++-----
 libgfortran/generated/findloc1_i2.c      |  48 ++++-----
 libgfortran/generated/findloc1_i4.c      |  48 ++++-----
 libgfortran/generated/findloc1_i8.c      |  48 ++++-----
 libgfortran/generated/findloc1_r10.c     |  48 ++++-----
 libgfortran/generated/findloc1_r16.c     |  48 ++++-----
 libgfortran/generated/findloc1_r17.c     |  48 ++++-----
 libgfortran/generated/findloc1_r4.c      |  48 ++++-----
 libgfortran/generated/findloc1_r8.c      |  48 ++++-----
 libgfortran/generated/findloc1_s1.c      |  48 ++++-----
 libgfortran/generated/findloc1_s4.c      |  48 ++++-----
 libgfortran/generated/iall_i1.c          |  48 ++++-----
 libgfortran/generated/iall_i16.c         |  48 ++++-----
 libgfortran/generated/iall_i2.c          |  48 ++++-----
 libgfortran/generated/iall_i4.c          |  48 ++++-----
 libgfortran/generated/iall_i8.c          |  48 ++++-----
 libgfortran/generated/iany_i1.c          |  48 ++++-----
 libgfortran/generated/iany_i16.c         |  48 ++++-----
 libgfortran/generated/iany_i2.c          |  48 ++++-----
 libgfortran/generated/iany_i4.c          |  48 ++++-----
 libgfortran/generated/iany_i8.c          |  48 ++++-----
 libgfortran/generated/iparity_i1.c       |  48 ++++-----
 libgfortran/generated/iparity_i16.c      |  48 ++++-----
 libgfortran/generated/iparity_i2.c       |  48 ++++-----
 libgfortran/generated/iparity_i4.c       |  48 ++++-----
 libgfortran/generated/iparity_i8.c       |  48 ++++-----
 libgfortran/generated/matmul_c10.c       | 165 +++++++++++++++++++------------
 libgfortran/generated/matmul_c16.c       | 165 +++++++++++++++++++------------
 libgfortran/generated/matmul_c17.c       | 165 +++++++++++++++++++------------
 libgfortran/generated/matmul_c4.c        | 165 +++++++++++++++++++------------
 libgfortran/generated/matmul_c8.c        | 165 +++++++++++++++++++------------
 libgfortran/generated/matmul_i1.c        | 165 +++++++++++++++++++------------
 libgfortran/generated/matmul_i16.c       | 165 +++++++++++++++++++------------
 libgfortran/generated/matmul_i2.c        | 165 +++++++++++++++++++------------
 libgfortran/generated/matmul_i4.c        | 165 +++++++++++++++++++------------
 libgfortran/generated/matmul_i8.c        | 165 +++++++++++++++++++------------
 libgfortran/generated/matmul_l16.c       |   6 +-
 libgfortran/generated/matmul_l4.c        |   6 +-
 libgfortran/generated/matmul_l8.c        |   6 +-
 libgfortran/generated/matmul_r10.c       | 165 +++++++++++++++++++------------
 libgfortran/generated/matmul_r16.c       | 165 +++++++++++++++++++------------
 libgfortran/generated/matmul_r17.c       | 165 +++++++++++++++++++------------
 libgfortran/generated/matmul_r4.c        | 165 +++++++++++++++++++------------
 libgfortran/generated/matmul_r8.c        | 165 +++++++++++++++++++------------
 libgfortran/generated/matmulavx128_c10.c |  66 ++++++++-----
 libgfortran/generated/matmulavx128_c16.c |  66 ++++++++-----
 libgfortran/generated/matmulavx128_c17.c |  66 ++++++++-----
 libgfortran/generated/matmulavx128_c4.c  |  66 ++++++++-----
 libgfortran/generated/matmulavx128_c8.c  |  66 ++++++++-----
 libgfortran/generated/matmulavx128_i1.c  |  66 ++++++++-----
 libgfortran/generated/matmulavx128_i16.c |  66 ++++++++-----
 libgfortran/generated/matmulavx128_i2.c  |  66 ++++++++-----
 libgfortran/generated/matmulavx128_i4.c  |  66 ++++++++-----
 libgfortran/generated/matmulavx128_i8.c  |  66 ++++++++-----
 libgfortran/generated/matmulavx128_r10.c |  66 ++++++++-----
 libgfortran/generated/matmulavx128_r16.c |  66 ++++++++-----
 libgfortran/generated/matmulavx128_r17.c |  66 ++++++++-----
 libgfortran/generated/matmulavx128_r4.c  |  66 ++++++++-----
 libgfortran/generated/matmulavx128_r8.c  |  66 ++++++++-----
 libgfortran/generated/maxloc1_16_i1.c    |  48 ++++-----
 libgfortran/generated/maxloc1_16_i16.c   |  48 ++++-----
 libgfortran/generated/maxloc1_16_i2.c    |  48 ++++-----
 libgfortran/generated/maxloc1_16_i4.c    |  48 ++++-----
 libgfortran/generated/maxloc1_16_i8.c    |  48 ++++-----
 libgfortran/generated/maxloc1_16_m1.c    |  48 ++++-----
 libgfortran/generated/maxloc1_16_m16.c   |  48 ++++-----
 libgfortran/generated/maxloc1_16_m2.c    |  48 ++++-----
 libgfortran/generated/maxloc1_16_m4.c    |  48 ++++-----
 libgfortran/generated/maxloc1_16_m8.c    |  48 ++++-----
 libgfortran/generated/maxloc1_16_r10.c   |  48 ++++-----
 libgfortran/generated/maxloc1_16_r16.c   |  48 ++++-----
 libgfortran/generated/maxloc1_16_r17.c   |  48 ++++-----
 libgfortran/generated/maxloc1_16_r4.c    |  48 ++++-----
 libgfortran/generated/maxloc1_16_r8.c    |  48 ++++-----
 libgfortran/generated/maxloc1_16_s1.c    |  48 ++++-----
 libgfortran/generated/maxloc1_16_s4.c    |  48 ++++-----
 libgfortran/generated/maxloc1_4_i1.c     |  48 ++++-----
 libgfortran/generated/maxloc1_4_i16.c    |  48 ++++-----
 libgfortran/generated/maxloc1_4_i2.c     |  48 ++++-----
 libgfortran/generated/maxloc1_4_i4.c     |  48 ++++-----
 libgfortran/generated/maxloc1_4_i8.c     |  48 ++++-----
 libgfortran/generated/maxloc1_4_m1.c     |  48 ++++-----
 libgfortran/generated/maxloc1_4_m16.c    |  48 ++++-----
 libgfortran/generated/maxloc1_4_m2.c     |  48 ++++-----
 libgfortran/generated/maxloc1_4_m4.c     |  48 ++++-----
 libgfortran/generated/maxloc1_4_m8.c     |  48 ++++-----
 libgfortran/generated/maxloc1_4_r10.c    |  48 ++++-----
 libgfortran/generated/maxloc1_4_r16.c    |  48 ++++-----
 libgfortran/generated/maxloc1_4_r17.c    |  48 ++++-----
 libgfortran/generated/maxloc1_4_r4.c     |  48 ++++-----
 libgfortran/generated/maxloc1_4_r8.c     |  48 ++++-----
 libgfortran/generated/maxloc1_4_s1.c     |  48 ++++-----
 libgfortran/generated/maxloc1_4_s4.c     |  48 ++++-----
 libgfortran/generated/maxloc1_8_i1.c     |  48 ++++-----
 libgfortran/generated/maxloc1_8_i16.c    |  48 ++++-----
 libgfortran/generated/maxloc1_8_i2.c     |  48 ++++-----
 libgfortran/generated/maxloc1_8_i4.c     |  48 ++++-----
 libgfortran/generated/maxloc1_8_i8.c     |  48 ++++-----
 libgfortran/generated/maxloc1_8_m1.c     |  48 ++++-----
 libgfortran/generated/maxloc1_8_m16.c    |  48 ++++-----
 libgfortran/generated/maxloc1_8_m2.c     |  48 ++++-----
 libgfortran/generated/maxloc1_8_m4.c     |  48 ++++-----
 libgfortran/generated/maxloc1_8_m8.c     |  48 ++++-----
 libgfortran/generated/maxloc1_8_r10.c    |  48 ++++-----
 libgfortran/generated/maxloc1_8_r16.c    |  48 ++++-----
 libgfortran/generated/maxloc1_8_r17.c    |  48 ++++-----
 libgfortran/generated/maxloc1_8_r4.c     |  48 ++++-----
 libgfortran/generated/maxloc1_8_r8.c     |  48 ++++-----
 libgfortran/generated/maxloc1_8_s1.c     |  48 ++++-----
 libgfortran/generated/maxloc1_8_s4.c     |  48 ++++-----
 libgfortran/generated/maxval1_s1.c       |  39 +++-----
 libgfortran/generated/maxval1_s4.c       |  39 +++-----
 libgfortran/generated/maxval_i1.c        |  48 ++++-----
 libgfortran/generated/maxval_i16.c       |  48 ++++-----
 libgfortran/generated/maxval_i2.c        |  48 ++++-----
 libgfortran/generated/maxval_i4.c        |  48 ++++-----
 libgfortran/generated/maxval_i8.c        |  48 ++++-----
 libgfortran/generated/maxval_m1.c        |  48 ++++-----
 libgfortran/generated/maxval_m16.c       |  48 ++++-----
 libgfortran/generated/maxval_m2.c        |  48 ++++-----
 libgfortran/generated/maxval_m4.c        |  48 ++++-----
 libgfortran/generated/maxval_m8.c        |  48 ++++-----
 libgfortran/generated/maxval_r10.c       |  48 ++++-----
 libgfortran/generated/maxval_r16.c       |  48 ++++-----
 libgfortran/generated/maxval_r17.c       |  48 ++++-----
 libgfortran/generated/maxval_r4.c        |  48 ++++-----
 libgfortran/generated/maxval_r8.c        |  48 ++++-----
 libgfortran/generated/minloc1_16_i1.c    |  48 ++++-----
 libgfortran/generated/minloc1_16_i16.c   |  48 ++++-----
 libgfortran/generated/minloc1_16_i2.c    |  48 ++++-----
 libgfortran/generated/minloc1_16_i4.c    |  48 ++++-----
 libgfortran/generated/minloc1_16_i8.c    |  48 ++++-----
 libgfortran/generated/minloc1_16_m1.c    |  48 ++++-----
 libgfortran/generated/minloc1_16_m16.c   |  48 ++++-----
 libgfortran/generated/minloc1_16_m2.c    |  48 ++++-----
 libgfortran/generated/minloc1_16_m4.c    |  48 ++++-----
 libgfortran/generated/minloc1_16_m8.c    |  48 ++++-----
 libgfortran/generated/minloc1_16_r10.c   |  48 ++++-----
 libgfortran/generated/minloc1_16_r16.c   |  48 ++++-----
 libgfortran/generated/minloc1_16_r17.c   |  48 ++++-----
 libgfortran/generated/minloc1_16_r4.c    |  48 ++++-----
 libgfortran/generated/minloc1_16_r8.c    |  48 ++++-----
 libgfortran/generated/minloc1_16_s1.c    |  48 ++++-----
 libgfortran/generated/minloc1_16_s4.c    |  48 ++++-----
 libgfortran/generated/minloc1_4_i1.c     |  48 ++++-----
 libgfortran/generated/minloc1_4_i16.c    |  48 ++++-----
 libgfortran/generated/minloc1_4_i2.c     |  48 ++++-----
 libgfortran/generated/minloc1_4_i4.c     |  48 ++++-----
 libgfortran/generated/minloc1_4_i8.c     |  48 ++++-----
 libgfortran/generated/minloc1_4_m1.c     |  48 ++++-----
 libgfortran/generated/minloc1_4_m16.c    |  48 ++++-----
 libgfortran/generated/minloc1_4_m2.c     |  48 ++++-----
 libgfortran/generated/minloc1_4_m4.c     |  48 ++++-----
 libgfortran/generated/minloc1_4_m8.c     |  48 ++++-----
 libgfortran/generated/minloc1_4_r10.c    |  48 ++++-----
 libgfortran/generated/minloc1_4_r16.c    |  48 ++++-----
 libgfortran/generated/minloc1_4_r17.c    |  48 ++++-----
 libgfortran/generated/minloc1_4_r4.c     |  48 ++++-----
 libgfortran/generated/minloc1_4_r8.c     |  48 ++++-----
 libgfortran/generated/minloc1_4_s1.c     |  48 ++++-----
 libgfortran/generated/minloc1_4_s4.c     |  48 ++++-----
 libgfortran/generated/minloc1_8_i1.c     |  48 ++++-----
 libgfortran/generated/minloc1_8_i16.c    |  48 ++++-----
 libgfortran/generated/minloc1_8_i2.c     |  48 ++++-----
 libgfortran/generated/minloc1_8_i4.c     |  48 ++++-----
 libgfortran/generated/minloc1_8_i8.c     |  48 ++++-----
 libgfortran/generated/minloc1_8_m1.c     |  48 ++++-----
 libgfortran/generated/minloc1_8_m16.c    |  48 ++++-----
 libgfortran/generated/minloc1_8_m2.c     |  48 ++++-----
 libgfortran/generated/minloc1_8_m4.c     |  48 ++++-----
 libgfortran/generated/minloc1_8_m8.c     |  48 ++++-----
 libgfortran/generated/minloc1_8_r10.c    |  48 ++++-----
 libgfortran/generated/minloc1_8_r16.c    |  48 ++++-----
 libgfortran/generated/minloc1_8_r17.c    |  48 ++++-----
 libgfortran/generated/minloc1_8_r4.c     |  48 ++++-----
 libgfortran/generated/minloc1_8_r8.c     |  48 ++++-----
 libgfortran/generated/minloc1_8_s1.c     |  48 ++++-----
 libgfortran/generated/minloc1_8_s4.c     |  48 ++++-----
 libgfortran/generated/minval1_s1.c       |  39 +++-----
 libgfortran/generated/minval1_s4.c       |  39 +++-----
 libgfortran/generated/minval_i1.c        |  48 ++++-----
 libgfortran/generated/minval_i16.c       |  48 ++++-----
 libgfortran/generated/minval_i2.c        |  48 ++++-----
 libgfortran/generated/minval_i4.c        |  48 ++++-----
 libgfortran/generated/minval_i8.c        |  48 ++++-----
 libgfortran/generated/minval_m1.c        |  48 ++++-----
 libgfortran/generated/minval_m16.c       |  48 ++++-----
 libgfortran/generated/minval_m2.c        |  48 ++++-----
 libgfortran/generated/minval_m4.c        |  48 ++++-----
 libgfortran/generated/minval_m8.c        |  48 ++++-----
 libgfortran/generated/minval_r10.c       |  48 ++++-----
 libgfortran/generated/minval_r16.c       |  48 ++++-----
 libgfortran/generated/minval_r17.c       |  48 ++++-----
 libgfortran/generated/minval_r4.c        |  48 ++++-----
 libgfortran/generated/minval_r8.c        |  48 ++++-----
 libgfortran/generated/norm2_r10.c        |  16 ++-
 libgfortran/generated/norm2_r16.c        |  16 ++-
 libgfortran/generated/norm2_r17.c        |  16 ++-
 libgfortran/generated/norm2_r4.c         |  16 ++-
 libgfortran/generated/norm2_r8.c         |  16 ++-
 libgfortran/generated/parity_l1.c        |  16 ++-
 libgfortran/generated/parity_l16.c       |  16 ++-
 libgfortran/generated/parity_l2.c        |  16 ++-
 libgfortran/generated/parity_l4.c        |  16 ++-
 libgfortran/generated/parity_l8.c        |  16 ++-
 libgfortran/generated/product_c10.c      |  48 ++++-----
 libgfortran/generated/product_c16.c      |  48 ++++-----
 libgfortran/generated/product_c17.c      |  48 ++++-----
 libgfortran/generated/product_c4.c       |  48 ++++-----
 libgfortran/generated/product_c8.c       |  48 ++++-----
 libgfortran/generated/product_i1.c       |  48 ++++-----
 libgfortran/generated/product_i16.c      |  48 ++++-----
 libgfortran/generated/product_i2.c       |  48 ++++-----
 libgfortran/generated/product_i4.c       |  48 ++++-----
 libgfortran/generated/product_i8.c       |  48 ++++-----
 libgfortran/generated/product_r10.c      |  48 ++++-----
 libgfortran/generated/product_r16.c      |  48 ++++-----
 libgfortran/generated/product_r17.c      |  48 ++++-----
 libgfortran/generated/product_r4.c       |  48 ++++-----
 libgfortran/generated/product_r8.c       |  48 ++++-----
 libgfortran/generated/reshape_c10.c      |   2 +-
 libgfortran/generated/reshape_c16.c      |   2 +-
 libgfortran/generated/reshape_c17.c      |   2 +-
 libgfortran/generated/reshape_c4.c       |   2 +-
 libgfortran/generated/reshape_c8.c       |   2 +-
 libgfortran/generated/reshape_i16.c      |   2 +-
 libgfortran/generated/reshape_i4.c       |   2 +-
 libgfortran/generated/reshape_i8.c       |   2 +-
 libgfortran/generated/reshape_r10.c      |   2 +-
 libgfortran/generated/reshape_r16.c      |   2 +-
 libgfortran/generated/reshape_r17.c      |   2 +-
 libgfortran/generated/reshape_r4.c       |   2 +-
 libgfortran/generated/reshape_r8.c       |   2 +-
 libgfortran/generated/spread_c10.c       |   3 +-
 libgfortran/generated/spread_c16.c       |   3 +-
 libgfortran/generated/spread_c17.c       |   3 +-
 libgfortran/generated/spread_c4.c        |   3 +-
 libgfortran/generated/spread_c8.c        |   3 +-
 libgfortran/generated/spread_i1.c        |   3 +-
 libgfortran/generated/spread_i16.c       |   3 +-
 libgfortran/generated/spread_i2.c        |   3 +-
 libgfortran/generated/spread_i4.c        |   3 +-
 libgfortran/generated/spread_i8.c        |   3 +-
 libgfortran/generated/spread_r10.c       |   3 +-
 libgfortran/generated/spread_r16.c       |   3 +-
 libgfortran/generated/spread_r17.c       |   3 +-
 libgfortran/generated/spread_r4.c        |   3 +-
 libgfortran/generated/spread_r8.c        |   3 +-
 libgfortran/generated/sum_c10.c          |  48 ++++-----
 libgfortran/generated/sum_c16.c          |  48 ++++-----
 libgfortran/generated/sum_c17.c          |  48 ++++-----
 libgfortran/generated/sum_c4.c           |  48 ++++-----
 libgfortran/generated/sum_c8.c           |  48 ++++-----
 libgfortran/generated/sum_i1.c           |  48 ++++-----
 libgfortran/generated/sum_i16.c          |  48 ++++-----
 libgfortran/generated/sum_i2.c           |  48 ++++-----
 libgfortran/generated/sum_i4.c           |  48 ++++-----
 libgfortran/generated/sum_i8.c           |  48 ++++-----
 libgfortran/generated/sum_r10.c          |  48 ++++-----
 libgfortran/generated/sum_r16.c          |  48 ++++-----
 libgfortran/generated/sum_r17.c          |  48 ++++-----
 libgfortran/generated/sum_r4.c           |  48 ++++-----
 libgfortran/generated/sum_r8.c           |  48 ++++-----
 308 files changed, 6052 insertions(+), 7769 deletions(-)

diff --git a/libgfortran/generated/all_l1.c b/libgfortran/generated/all_l1.c
index fafdb05209e2..cf66314586e2 100644
--- a/libgfortran/generated/all_l1.c
+++ b/libgfortran/generated/all_l1.c
@@ -83,25 +83,21 @@ all_l1 (gfc_array_l1 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
         {
-          if (n == 0)
-            str = 1;
-          else
-            str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+          cnt = cnt * extent[n];
         }
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_LOGICAL_1));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_LOGICAL_1));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/all_l16.c b/libgfortran/generated/all_l16.c
index 3a070ecd5413..6070a2b8f797 100644
--- a/libgfortran/generated/all_l16.c
+++ b/libgfortran/generated/all_l16.c
@@ -83,25 +83,21 @@ all_l16 (gfc_array_l16 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
         {
-          if (n == 0)
-            str = 1;
-          else
-            str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+          cnt = cnt * extent[n];
         }
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_LOGICAL_16));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_LOGICAL_16));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/all_l2.c b/libgfortran/generated/all_l2.c
index 6d6de77b5d99..9560ec8ee226 100644
--- a/libgfortran/generated/all_l2.c
+++ b/libgfortran/generated/all_l2.c
@@ -83,25 +83,21 @@ all_l2 (gfc_array_l2 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
         {
-          if (n == 0)
-            str = 1;
-          else
-            str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+          cnt = cnt * extent[n];
         }
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_LOGICAL_2));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_LOGICAL_2));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/all_l4.c b/libgfortran/generated/all_l4.c
index aedb1db2b907..a802574800f4 100644
--- a/libgfortran/generated/all_l4.c
+++ b/libgfortran/generated/all_l4.c
@@ -83,25 +83,21 @@ all_l4 (gfc_array_l4 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
         {
-          if (n == 0)
-            str = 1;
-          else
-            str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+          cnt = cnt * extent[n];
         }
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_LOGICAL_4));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_LOGICAL_4));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/all_l8.c b/libgfortran/generated/all_l8.c
index 1a7d8c025b06..d5ce92882ed2 100644
--- a/libgfortran/generated/all_l8.c
+++ b/libgfortran/generated/all_l8.c
@@ -83,25 +83,21 @@ all_l8 (gfc_array_l8 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
         {
-          if (n == 0)
-            str = 1;
-          else
-            str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+          cnt = cnt * extent[n];
         }
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_LOGICAL_8));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_LOGICAL_8));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/any_l1.c b/libgfortran/generated/any_l1.c
index fd9ec847cf51..83c89458b3da 100644
--- a/libgfortran/generated/any_l1.c
+++ b/libgfortran/generated/any_l1.c
@@ -83,25 +83,21 @@ any_l1 (gfc_array_l1 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
         {
-          if (n == 0)
-            str = 1;
-          else
-            str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+          cnt = cnt * extent[n];
         }
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_LOGICAL_1));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_LOGICAL_1));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/any_l16.c b/libgfortran/generated/any_l16.c
index aa803868216c..f56abbe2e55e 100644
--- a/libgfortran/generated/any_l16.c
+++ b/libgfortran/generated/any_l16.c
@@ -83,25 +83,21 @@ any_l16 (gfc_array_l16 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
         {
-          if (n == 0)
-            str = 1;
-          else
-            str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+          cnt = cnt * extent[n];
         }
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_LOGICAL_16));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_LOGICAL_16));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/any_l2.c b/libgfortran/generated/any_l2.c
index 9dbe864b0e95..da5cd0a98192 100644
--- a/libgfortran/generated/any_l2.c
+++ b/libgfortran/generated/any_l2.c
@@ -83,25 +83,21 @@ any_l2 (gfc_array_l2 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
         {
-          if (n == 0)
-            str = 1;
-          else
-            str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+          cnt = cnt * extent[n];
         }
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_LOGICAL_2));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_LOGICAL_2));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/any_l4.c b/libgfortran/generated/any_l4.c
index dfdb5e697276..72eccdbbc20d 100644
--- a/libgfortran/generated/any_l4.c
+++ b/libgfortran/generated/any_l4.c
@@ -83,25 +83,21 @@ any_l4 (gfc_array_l4 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
         {
-          if (n == 0)
-            str = 1;
-          else
-            str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+          cnt = cnt * extent[n];
         }
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_LOGICAL_4));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_LOGICAL_4));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/any_l8.c b/libgfortran/generated/any_l8.c
index 8ff8f563eca8..25eb313a042b 100644
--- a/libgfortran/generated/any_l8.c
+++ b/libgfortran/generated/any_l8.c
@@ -83,25 +83,21 @@ any_l8 (gfc_array_l8 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
         {
-          if (n == 0)
-            str = 1;
-          else
-            str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+          cnt = cnt * extent[n];
         }
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_LOGICAL_8));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_LOGICAL_8));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/count_16_l.c b/libgfortran/generated/count_16_l.c
index d514c1f5551f..28c604e38311 100644
--- a/libgfortran/generated/count_16_l.c
+++ b/libgfortran/generated/count_16_l.c
@@ -83,25 +83,21 @@ count_16_l (gfc_array_i16 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
         {
-          if (n == 0)
-            str = 1;
-          else
-            str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+          cnt = cnt * extent[n];
         }
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_16));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_16));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/count_1_l.c b/libgfortran/generated/count_1_l.c
index f60e49c492f7..2675346e6d92 100644
--- a/libgfortran/generated/count_1_l.c
+++ b/libgfortran/generated/count_1_l.c
@@ -83,25 +83,21 @@ count_1_l (gfc_array_i1 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
         {
-          if (n == 0)
-            str = 1;
-          else
-            str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+          cnt = cnt * extent[n];
         }
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_1));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_1));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/count_2_l.c b/libgfortran/generated/count_2_l.c
index 75c676bc9e62..d77bedd9d4d1 100644
--- a/libgfortran/generated/count_2_l.c
+++ b/libgfortran/generated/count_2_l.c
@@ -83,25 +83,21 @@ count_2_l (gfc_array_i2 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
         {
-          if (n == 0)
-            str = 1;
-          else
-            str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+          cnt = cnt * extent[n];
         }
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_2));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_2));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/count_4_l.c b/libgfortran/generated/count_4_l.c
index c03ca930736b..08c2cbd1d9e6 100644
--- a/libgfortran/generated/count_4_l.c
+++ b/libgfortran/generated/count_4_l.c
@@ -83,25 +83,21 @@ count_4_l (gfc_array_i4 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
         {
-          if (n == 0)
-            str = 1;
-          else
-            str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+          cnt = cnt * extent[n];
         }
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_4));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_4));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/count_8_l.c b/libgfortran/generated/count_8_l.c
index e4ab3bbc9b27..288f6111a7d2 100644
--- a/libgfortran/generated/count_8_l.c
+++ b/libgfortran/generated/count_8_l.c
@@ -83,25 +83,21 @@ count_8_l (gfc_array_i8 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
         {
-          if (n == 0)
-            str = 1;
-          else
-            str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+          cnt = cnt * extent[n];
         }
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_8));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_8));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/cshift0_c10.c b/libgfortran/generated/cshift0_c10.c
index 3908c5b693f3..9b8d57303430 100644
--- a/libgfortran/generated/cshift0_c10.c
+++ b/libgfortran/generated/cshift0_c10.c
@@ -47,6 +47,7 @@ cshift0_c10 (gfc_array_c10 *ret, const gfc_array_c10 *array, ptrdiff_t shift,
 
   index_type count[GFC_MAX_DIMENSIONS];
   index_type extent[GFC_MAX_DIMENSIONS];
+  index_type contiguous_extent;
   index_type dim;
   index_type len;
   index_type n;
@@ -66,31 +67,34 @@ cshift0_c10 (gfc_array_c10 *ret, const gfc_array_c10 *array, ptrdiff_t shift,
   soffset = 1;
   len = 0;
 
-  r_ex = 1;
-  a_ex = 1;
+  r_ex = sizeof (GFC_COMPLEX_10);
+  a_ex = sizeof (GFC_COMPLEX_10);
 
   if (which > 0)
     {
       /* Test if both ret and array are contiguous.  */
       do_blocked = true;
+      contiguous_extent = 1;
       dim = GFC_DESCRIPTOR_RANK (array);
       for (n = 0; n < dim; n ++)
 	{
 	  index_type rs, as;
-	  rs = GFC_DESCRIPTOR_STRIDE (ret, n);
+	  rs = GFC_DESCRIPTOR_STRIDE_BYTES (ret, n);
 	  if (rs != r_ex)
 	    {
 	      do_blocked = false;
 	      break;
 	    }
-	  as = GFC_DESCRIPTOR_STRIDE (array, n);
+	  as = GFC_DESCRIPTOR_STRIDE_BYTES (array, n);
 	  if (as != a_ex)
 	    {
 	      do_blocked = false;
 	      break;
 	    }
+	  index_type extent = GFC_DESCRIPTOR_EXTENT (array, n);
 	  r_ex *= GFC_DESCRIPTOR_EXTENT (ret, n);
-	  a_ex *= GFC_DESCRIPTOR_EXTENT (array, n);
+	  a_ex *= extent;
+	  contiguous_extent = extent;
 	}
     }
   else
@@ -115,9 +119,8 @@ cshift0_c10 (gfc_array_c10 *ret, const gfc_array_c10 *array, ptrdiff_t shift,
       rstride[0] = sizeof (GFC_COMPLEX_10);
       roffset = sizeof (GFC_COMPLEX_10);
       soffset = sizeof (GFC_COMPLEX_10);
-      len = GFC_DESCRIPTOR_STRIDE(array, which)
-	* GFC_DESCRIPTOR_EXTENT(array, which);      
-      shift *= GFC_DESCRIPTOR_STRIDE(array, which);
+      len = contiguous_extent * GFC_DESCRIPTOR_EXTENT(array, which);
+      shift *= contiguous_extent;
       for (dim = which + 1; dim < GFC_DESCRIPTOR_RANK (array); dim++)
 	{
 	  count[n] = 0;
diff --git a/libgfortran/generated/cshift0_c16.c b/libgfortran/generated/cshift0_c16.c
index 962c009f3568..d89b538c7476 100644
--- a/libgfortran/generated/cshift0_c16.c
+++ b/libgfortran/generated/cshift0_c16.c
@@ -47,6 +47,7 @@ cshift0_c16 (gfc_array_c16 *ret, const gfc_array_c16 *array, ptrdiff_t shift,
 
   index_type count[GFC_MAX_DIMENSIONS];
   index_type extent[GFC_MAX_DIMENSIONS];
+  index_type contiguous_extent;
   index_type dim;
   index_type len;
   index_type n;
@@ -66,31 +67,34 @@ cshift0_c16 (gfc_array_c16 *ret, const gfc_array_c16 *array, ptrdiff_t shift,
   soffset = 1;
   len = 0;
 
-  r_ex = 1;
-  a_ex = 1;
+  r_ex = sizeof (GFC_COMPLEX_16);
+  a_ex = sizeof (GFC_COMPLEX_16);
 
   if (which > 0)
     {
       /* Test if both ret and array are contiguous.  */
       do_blocked = true;
+      contiguous_extent = 1;
       dim = GFC_DESCRIPTOR_RANK (array);
       for (n = 0; n < dim; n ++)
 	{
 	  index_type rs, as;
-	  rs = GFC_DESCRIPTOR_STRIDE (ret, n);
+	  rs = GFC_DESCRIPTOR_STRIDE_BYTES (ret, n);
 	  if (rs != r_ex)
 	    {
 	      do_blocked = false;
 	      break;
 	    }
-	  as = GFC_DESCRIPTOR_STRIDE (array, n);
+	  as = GFC_DESCRIPTOR_STRIDE_BYTES (array, n);
 	  if (as != a_ex)
 	    {
 	      do_blocked = false;
 	      break;
 	    }
+	  index_type extent = GFC_DESCRIPTOR_EXTENT (array, n);
 	  r_ex *= GFC_DESCRIPTOR_EXTENT (ret, n);
-	  a_ex *= GFC_DESCRIPTOR_EXTENT (array, n);
+	  a_ex *= extent;
+	  contiguous_extent = extent;
 	}
     }
   else
@@ -115,9 +119,8 @@ cshift0_c16 (gfc_array_c16 *ret, const gfc_array_c16 *array, ptrdiff_t shift,
       rstride[0] = sizeof (GFC_COMPLEX_16);
       roffset = sizeof (GFC_COMPLEX_16);
       soffset = sizeof (GFC_COMPLEX_16);
-      len = GFC_DESCRIPTOR_STRIDE(array, which)
-	* GFC_DESCRIPTOR_EXTENT(array, which);      
-      shift *= GFC_DESCRIPTOR_STRIDE(array, which);
+      len = contiguous_extent * GFC_DESCRIPTOR_EXTENT(array, which);
+      shift *= contiguous_extent;
       for (dim = which + 1; dim < GFC_DESCRIPTOR_RANK (array); dim++)
 	{
 	  count[n] = 0;
diff --git a/libgfortran/generated/cshift0_c17.c b/libgfortran/generated/cshift0_c17.c
index ebe965efcb5d..a47be7d04ade 100644
--- a/libgfortran/generated/cshift0_c17.c
+++ b/libgfortran/generated/cshift0_c17.c
@@ -47,6 +47,7 @@ cshift0_c17 (gfc_array_c17 *ret, const gfc_array_c17 *array, ptrdiff_t shift,
 
   index_type count[GFC_MAX_DIMENSIONS];
   index_type extent[GFC_MAX_DIMENSIONS];
+  index_type contiguous_extent;
   index_type dim;
   index_type len;
   index_type n;
@@ -66,31 +67,34 @@ cshift0_c17 (gfc_array_c17 *ret, const gfc_array_c17 *array, ptrdiff_t shift,
   soffset = 1;
   len = 0;
 
-  r_ex = 1;
-  a_ex = 1;
+  r_ex = sizeof (GFC_COMPLEX_17);
+  a_ex = sizeof (GFC_COMPLEX_17);
 
   if (which > 0)
     {
       /* Test if both ret and array are contiguous.  */
       do_blocked = true;
+      contiguous_extent = 1;
       dim = GFC_DESCRIPTOR_RANK (array);
       for (n = 0; n < dim; n ++)
 	{
 	  index_type rs, as;
-	  rs = GFC_DESCRIPTOR_STRIDE (ret, n);
+	  rs = GFC_DESCRIPTOR_STRIDE_BYTES (ret, n);
 	  if (rs != r_ex)
 	    {
 	      do_blocked = false;
 	      break;
 	    }
-	  as = GFC_DESCRIPTOR_STRIDE (array, n);
+	  as = GFC_DESCRIPTOR_STRIDE_BYTES (array, n);
 	  if (as != a_ex)
 	    {
 	      do_blocked = false;
 	      break;
 	    }
+	  index_type extent = GFC_DESCRIPTOR_EXTENT (array, n);
 	  r_ex *= GFC_DESCRIPTOR_EXTENT (ret, n);
-	  a_ex *= GFC_DESCRIPTOR_EXTENT (array, n);
+	  a_ex *= extent;
+	  contiguous_extent = extent;
 	}
     }
   else
@@ -115,9 +119,8 @@ cshift0_c17 (gfc_array_c17 *ret, const gfc_array_c17 *array, ptrdiff_t shift,
       rstride[0] = sizeof (GFC_COMPLEX_17);
       roffset = sizeof (GFC_COMPLEX_17);
       soffset = sizeof (GFC_COMPLEX_17);
-      len = GFC_DESCRIPTOR_STRIDE(array, which)
-	* GFC_DESCRIPTOR_EXTENT(array, which);      
-      shift *= GFC_DESCRIPTOR_STRIDE(array, which);
+      len = contiguous_extent * GFC_DESCRIPTOR_EXTENT(array, which);
+      shift *= contiguous_extent;
       for (dim = which + 1; dim < GFC_DESCRIPTOR_RANK (array); dim++)
 	{
 	  count[n] = 0;
diff --git a/libgfortran/generated/cshift0_c4.c b/libgfortran/generated/cshift0_c4.c
index 71758c0aed18..5120d7270a46 100644
--- a/libgfortran/generated/cshift0_c4.c
+++ b/libgfortran/generated/cshift0_c4.c
@@ -47,6 +47,7 @@ cshift0_c4 (gfc_array_c4 *ret, const gfc_array_c4 *array, ptrdiff_t shift,
 
   index_type count[GFC_MAX_DIMENSIONS];
   index_type extent[GFC_MAX_DIMENSIONS];
+  index_type contiguous_extent;
   index_type dim;
   index_type len;
   index_type n;
@@ -66,31 +67,34 @@ cshift0_c4 (gfc_array_c4 *ret, const gfc_array_c4 *array, ptrdiff_t shift,
   soffset = 1;
   len = 0;
 
-  r_ex = 1;
-  a_ex = 1;
+  r_ex = sizeof (GFC_COMPLEX_4);
+  a_ex = sizeof (GFC_COMPLEX_4);
 
   if (which > 0)
     {
       /* Test if both ret and array are contiguous.  */
       do_blocked = true;
+      contiguous_extent = 1;
       dim = GFC_DESCRIPTOR_RANK (array);
       for (n = 0; n < dim; n ++)
 	{
 	  index_type rs, as;
-	  rs = GFC_DESCRIPTOR_STRIDE (ret, n);
+	  rs = GFC_DESCRIPTOR_STRIDE_BYTES (ret, n);
 	  if (rs != r_ex)
 	    {
 	      do_blocked = false;
 	      break;
 	    }
-	  as = GFC_DESCRIPTOR_STRIDE (array, n);
+	  as = GFC_DESCRIPTOR_STRIDE_BYTES (array, n);
 	  if (as != a_ex)
 	    {
 	      do_blocked = false;
 	      break;
 	    }
+	  index_type extent = GFC_DESCRIPTOR_EXTENT (array, n);
 	  r_ex *= GFC_DESCRIPTOR_EXTENT (ret, n);
-	  a_ex *= GFC_DESCRIPTOR_EXTENT (array, n);
+	  a_ex *= extent;
+	  contiguous_extent = extent;
 	}
     }
   else
@@ -115,9 +119,8 @@ cshift0_c4 (gfc_array_c4 *ret, const gfc_array_c4 *array, ptrdiff_t shift,
       rstride[0] = sizeof (GFC_COMPLEX_4);
       roffset = sizeof (GFC_COMPLEX_4);
       soffset = sizeof (GFC_COMPLEX_4);
-      len = GFC_DESCRIPTOR_STRIDE(array, which)
-	* GFC_DESCRIPTOR_EXTENT(array, which);      
-      shift *= GFC_DESCRIPTOR_STRIDE(array, which);
+      len = contiguous_extent * GFC_DESCRIPTOR_EXTENT(array, which);
+      shift *= contiguous_extent;
       for (dim = which + 1; dim < GFC_DESCRIPTOR_RANK (array); dim++)
 	{
 	  count[n] = 0;
diff --git a/libgfortran/generated/cshift0_c8.c b/libgfortran/generated/cshift0_c8.c
index 1b9bff6e7627..116ca9317887 100644
--- a/libgfortran/generated/cshift0_c8.c
+++ b/libgfortran/generated/cshift0_c8.c
@@ -47,6 +47,7 @@ cshift0_c8 (gfc_array_c8 *ret, const gfc_array_c8 *array, ptrdiff_t shift,
 
   index_type count[GFC_MAX_DIMENSIONS];
   index_type extent[GFC_MAX_DIMENSIONS];
+  index_type contiguous_extent;
   index_type dim;
   index_type len;
   index_type n;
@@ -66,31 +67,34 @@ cshift0_c8 (gfc_array_c8 *ret, const gfc_array_c8 *array, ptrdiff_t shift,
   soffset = 1;
   len = 0;
 
-  r_ex = 1;
-  a_ex = 1;
+  r_ex = sizeof (GFC_COMPLEX_8);
+  a_ex = sizeof (GFC_COMPLEX_8);
 
   if (which > 0)
     {
       /* Test if both ret and array are contiguous.  */
       do_blocked = true;
+      contiguous_extent = 1;
       dim = GFC_DESCRIPTOR_RANK (array);
       for (n = 0; n < dim; n ++)
 	{
 	  index_type rs, as;
-	  rs = GFC_DESCRIPTOR_STRIDE (ret, n);
+	  rs = GFC_DESCRIPTOR_STRIDE_BYTES (ret, n);
 	  if (rs != r_ex)
 	    {
 	      do_blocked = false;
 	      break;
 	    }
-	  as = GFC_DESCRIPTOR_STRIDE (array, n);
+	  as = GFC_DESCRIPTOR_STRIDE_BYTES (array, n);
 	  if (as != a_ex)
 	    {
 	      do_blocked = false;
 	      break;
 	    }
+	  index_type extent = GFC_DESCRIPTOR_EXTENT (array, n);
 	  r_ex *= GFC_DESCRIPTOR_EXTENT (ret, n);
-	  a_ex *= GFC_DESCRIPTOR_EXTENT (array, n);
+	  a_ex *= extent;
+	  contiguous_extent = extent;
 	}
     }
   else
@@ -115,9 +119,8 @@ cshift0_c8 (gfc_array_c8 *ret, const gfc_array_c8 *array, ptrdiff_t shift,
       rstride[0] = sizeof (GFC_COMPLEX_8);
       roffset = sizeof (GFC_COMPLEX_8);
       soffset = sizeof (GFC_COMPLEX_8);
-      len = GFC_DESCRIPTOR_STRIDE(array, which)
-	* GFC_DESCRIPTOR_EXTENT(array, which);      
-      shift *= GFC_DESCRIPTOR_STRIDE(array, which);
+      len = contiguous_extent * GFC_DESCRIPTOR_EXTENT(array, which);
+      shift *= contiguous_extent;
       for (dim = which + 1; dim < GFC_DESCRIPTOR_RANK (array); dim++)
 	{
 	  count[n] = 0;
diff --git a/libgfortran/generated/cshift0_i1.c b/libgfortran/generated/cshift0_i1.c
index c34a72e8612f..9b2bc94afd57 100644
--- a/libgfortran/generated/cshift0_i1.c
+++ b/libgfortran/generated/cshift0_i1.c
@@ -47,6 +47,7 @@ cshift0_i1 (gfc_array_i1 *ret, const gfc_array_i1 *array, ptrdiff_t shift,
 
   index_type count[GFC_MAX_DIMENSIONS];
   index_type extent[GFC_MAX_DIMENSIONS];
+  index_type contiguous_extent;
   index_type dim;
   index_type len;
   index_type n;
@@ -66,31 +67,34 @@ cshift0_i1 (gfc_array_i1 *ret, const gfc_array_i1 *array, ptrdiff_t shift,
   soffset = 1;
   len = 0;
 
-  r_ex = 1;
-  a_ex = 1;
+  r_ex = sizeof (GFC_INTEGER_1);
+  a_ex = sizeof (GFC_INTEGER_1);
 
   if (which > 0)
     {
       /* Test if both ret and array are contiguous.  */
       do_blocked = true;
+      contiguous_extent = 1;
       dim = GFC_DESCRIPTOR_RANK (array);
       for (n = 0; n < dim; n ++)
 	{
 	  index_type rs, as;
-	  rs = GFC_DESCRIPTOR_STRIDE (ret, n);
+	  rs = GFC_DESCRIPTOR_STRIDE_BYTES (ret, n);
 	  if (rs != r_ex)
 	    {
 	      do_blocked = false;
 	      break;
 	    }
-	  as = GFC_DESCRIPTOR_STRIDE (array, n);
+	  as = GFC_DESCRIPTOR_STRIDE_BYTES (array, n);
 	  if (as != a_ex)
 	    {
 	      do_blocked = false;
 	      break;
 	    }
+	  index_type extent = GFC_DESCRIPTOR_EXTENT (array, n);
 	  r_ex *= GFC_DESCRIPTOR_EXTENT (ret, n);
-	  a_ex *= GFC_DESCRIPTOR_EXTENT (array, n);
+	  a_ex *= extent;
+	  contiguous_extent = extent;
 	}
     }
   else
@@ -115,9 +119,8 @@ cshift0_i1 (gfc_array_i1 *ret, const gfc_array_i1 *array, ptrdiff_t shift,
       rstride[0] = sizeof (GFC_INTEGER_1);
       roffset = sizeof (GFC_INTEGER_1);
       soffset = sizeof (GFC_INTEGER_1);
-      len = GFC_DESCRIPTOR_STRIDE(array, which)
-	* GFC_DESCRIPTOR_EXTENT(array, which);      
-      shift *= GFC_DESCRIPTOR_STRIDE(array, which);
+      len = contiguous_extent * GFC_DESCRIPTOR_EXTENT(array, which);
+      shift *= contiguous_extent;
       for (dim = which + 1; dim < GFC_DESCRIPTOR_RANK (array); dim++)
 	{
 	  count[n] = 0;
diff --git a/libgfortran/generated/cshift0_i16.c b/libgfortran/generated/cshift0_i16.c
index 01b502f5c538..26a7bbf8dadd 100644
--- a/libgfortran/generated/cshift0_i16.c
+++ b/libgfortran/generated/cshift0_i16.c
@@ -47,6 +47,7 @@ cshift0_i16 (gfc_array_i16 *ret, const gfc_array_i16 *array, ptrdiff_t shift,
 
   index_type count[GFC_MAX_DIMENSIONS];
   index_type extent[GFC_MAX_DIMENSIONS];
+  index_type contiguous_extent;
   index_type dim;
   index_type len;
   index_type n;
@@ -66,31 +67,34 @@ cshift0_i16 (gfc_array_i16 *ret, const gfc_array_i16 *array, ptrdiff_t shift,
   soffset = 1;
   len = 0;
 
-  r_ex = 1;
-  a_ex = 1;
+  r_ex = sizeof (GFC_INTEGER_16);
+  a_ex = sizeof (GFC_INTEGER_16);
 
   if (which > 0)
     {
       /* Test if both ret and array are contiguous.  */
       do_blocked = true;
+      contiguous_extent = 1;
       dim = GFC_DESCRIPTOR_RANK (array);
       for (n = 0; n < dim; n ++)
 	{
 	  index_type rs, as;
-	  rs = GFC_DESCRIPTOR_STRIDE (ret, n);
+	  rs = GFC_DESCRIPTOR_STRIDE_BYTES (ret, n);
 	  if (rs != r_ex)
 	    {
 	      do_blocked = false;
 	      break;
 	    }
-	  as = GFC_DESCRIPTOR_STRIDE (array, n);
+	  as = GFC_DESCRIPTOR_STRIDE_BYTES (array, n);
 	  if (as != a_ex)
 	    {
 	      do_blocked = false;
 	      break;
 	    }
+	  index_type extent = GFC_DESCRIPTOR_EXTENT (array, n);
 	  r_ex *= GFC_DESCRIPTOR_EXTENT (ret, n);
-	  a_ex *= GFC_DESCRIPTOR_EXTENT (array, n);
+	  a_ex *= extent;
+	  contiguous_extent = extent;
 	}
     }
   else
@@ -115,9 +119,8 @@ cshift0_i16 (gfc_array_i16 *ret, const gfc_array_i16 *array, ptrdiff_t shift,
       rstride[0] = sizeof (GFC_INTEGER_16);
       roffset = sizeof (GFC_INTEGER_16);
       soffset = sizeof (GFC_INTEGER_16);
-      len = GFC_DESCRIPTOR_STRIDE(array, which)
-	* GFC_DESCRIPTOR_EXTENT(array, which);      
-      shift *= GFC_DESCRIPTOR_STRIDE(array, which);
+      len = contiguous_extent * GFC_DESCRIPTOR_EXTENT(array, which);
+      shift *= contiguous_extent;
       for (dim = which + 1; dim < GFC_DESCRIPTOR_RANK (array); dim++)
 	{
 	  count[n] = 0;
diff --git a/libgfortran/generated/cshift0_i2.c b/libgfortran/generated/cshift0_i2.c
index 743cb6ececda..9e72ed23fbff 100644
--- a/libgfortran/generated/cshift0_i2.c
+++ b/libgfortran/generated/cshift0_i2.c
@@ -47,6 +47,7 @@ cshift0_i2 (gfc_array_i2 *ret, const gfc_array_i2 *array, ptrdiff_t shift,
 
   index_type count[GFC_MAX_DIMENSIONS];
   index_type extent[GFC_MAX_DIMENSIONS];
+  index_type contiguous_extent;
   index_type dim;
   index_type len;
   index_type n;
@@ -66,31 +67,34 @@ cshift0_i2 (gfc_array_i2 *ret, const gfc_array_i2 *array, ptrdiff_t shift,
   soffset = 1;
   len = 0;
 
-  r_ex = 1;
-  a_ex = 1;
+  r_ex = sizeof (GFC_INTEGER_2);
+  a_ex = sizeof (GFC_INTEGER_2);
 
   if (which > 0)
     {
       /* Test if both ret and array are contiguous.  */
       do_blocked = true;
+      contiguous_extent = 1;
       dim = GFC_DESCRIPTOR_RANK (array);
       for (n = 0; n < dim; n ++)
 	{
 	  index_type rs, as;
-	  rs = GFC_DESCRIPTOR_STRIDE (ret, n);
+	  rs = GFC_DESCRIPTOR_STRIDE_BYTES (ret, n);
 	  if (rs != r_ex)
 	    {
 	      do_blocked = false;
 	      break;
 	    }
-	  as = GFC_DESCRIPTOR_STRIDE (array, n);
+	  as = GFC_DESCRIPTOR_STRIDE_BYTES (array, n);
 	  if (as != a_ex)
 	    {
 	      do_blocked = false;
 	      break;
 	    }
+	  index_type extent = GFC_DESCRIPTOR_EXTENT (array, n);
 	  r_ex *= GFC_DESCRIPTOR_EXTENT (ret, n);
-	  a_ex *= GFC_DESCRIPTOR_EXTENT (array, n);
+	  a_ex *= extent;
+	  contiguous_extent = extent;
 	}
     }
   else
@@ -115,9 +119,8 @@ cshift0_i2 (gfc_array_i2 *ret, const gfc_array_i2 *array, ptrdiff_t shift,
       rstride[0] = sizeof (GFC_INTEGER_2);
       roffset = sizeof (GFC_INTEGER_2);
       soffset = sizeof (GFC_INTEGER_2);
-      len = GFC_DESCRIPTOR_STRIDE(array, which)
-	* GFC_DESCRIPTOR_EXTENT(array, which);      
-      shift *= GFC_DESCRIPTOR_STRIDE(array, which);
+      len = contiguous_extent * GFC_DESCRIPTOR_EXTENT(array, which);
+      shift *= contiguous_extent;
       for (dim = which + 1; dim < GFC_DESCRIPTOR_RANK (array); dim++)
 	{
 	  count[n] = 0;
diff --git a/libgfortran/generated/cshift0_i4.c b/libgfortran/generated/cshift0_i4.c
index ccd1d2424a25..7843d3d96aca 100644
--- a/libgfortran/generated/cshift0_i4.c
+++ b/libgfortran/generated/cshift0_i4.c
@@ -47,6 +47,7 @@ cshift0_i4 (gfc_array_i4 *ret, const gfc_array_i4 *array, ptrdiff_t shift,
 
   index_type count[GFC_MAX_DIMENSIONS];
   index_type extent[GFC_MAX_DIMENSIONS];
+  index_type contiguous_extent;
   index_type dim;
   index_type len;
   index_type n;
@@ -66,31 +67,34 @@ cshift0_i4 (gfc_array_i4 *ret, const gfc_array_i4 *array, ptrdiff_t shift,
   soffset = 1;
   len = 0;
 
-  r_ex = 1;
-  a_ex = 1;
+  r_ex = sizeof (GFC_INTEGER_4);
+  a_ex = sizeof (GFC_INTEGER_4);
 
   if (which > 0)
     {
       /* Test if both ret and array are contiguous.  */
       do_blocked = true;
+      contiguous_extent = 1;
       dim = GFC_DESCRIPTOR_RANK (array);
       for (n = 0; n < dim; n ++)
 	{
 	  index_type rs, as;
-	  rs = GFC_DESCRIPTOR_STRIDE (ret, n);
+	  rs = GFC_DESCRIPTOR_STRIDE_BYTES (ret, n);
 	  if (rs != r_ex)
 	    {
 	      do_blocked = false;
 	      break;
 	    }
-	  as = GFC_DESCRIPTOR_STRIDE (array, n);
+	  as = GFC_DESCRIPTOR_STRIDE_BYTES (array, n);
 	  if (as != a_ex)
 	    {
 	      do_blocked = false;
 	      break;
 	    }
+	  index_type extent = GFC_DESCRIPTOR_EXTENT (array, n);
 	  r_ex *= GFC_DESCRIPTOR_EXTENT (ret, n);
-	  a_ex *= GFC_DESCRIPTOR_EXTENT (array, n);
+	  a_ex *= extent;
+	  contiguous_extent = extent;
 	}
     }
   else
@@ -115,9 +119,8 @@ cshift0_i4 (gfc_array_i4 *ret, const gfc_array_i4 *array, ptrdiff_t shift,
       rstride[0] = sizeof (GFC_INTEGER_4);
       roffset = sizeof (GFC_INTEGER_4);
       soffset = sizeof (GFC_INTEGER_4);
-      len = GFC_DESCRIPTOR_STRIDE(array, which)
-	* GFC_DESCRIPTOR_EXTENT(array, which);      
-      shift *= GFC_DESCRIPTOR_STRIDE(array, which);
+      len = contiguous_extent * GFC_DESCRIPTOR_EXTENT(array, which);
+      shift *= contiguous_extent;
       for (dim = which + 1; dim < GFC_DESCRIPTOR_RANK (array); dim++)
 	{
 	  count[n] = 0;
diff --git a/libgfortran/generated/cshift0_i8.c b/libgfortran/generated/cshift0_i8.c
index defbb1149c08..b0230b777970 100644
--- a/libgfortran/generated/cshift0_i8.c
+++ b/libgfortran/generated/cshift0_i8.c
@@ -47,6 +47,7 @@ cshift0_i8 (gfc_array_i8 *ret, const gfc_array_i8 *array, ptrdiff_t shift,
 
   index_type count[GFC_MAX_DIMENSIONS];
   index_type extent[GFC_MAX_DIMENSIONS];
+  index_type contiguous_extent;
   index_type dim;
   index_type len;
   index_type n;
@@ -66,31 +67,34 @@ cshift0_i8 (gfc_array_i8 *ret, const gfc_array_i8 *array, ptrdiff_t shift,
   soffset = 1;
   len = 0;
 
-  r_ex = 1;
-  a_ex = 1;
+  r_ex = sizeof (GFC_INTEGER_8);
+  a_ex = sizeof (GFC_INTEGER_8);
 
   if (which > 0)
     {
       /* Test if both ret and array are contiguous.  */
       do_blocked = true;
+      contiguous_extent = 1;
       dim = GFC_DESCRIPTOR_RANK (array);
       for (n = 0; n < dim; n ++)
 	{
 	  index_type rs, as;
-	  rs = GFC_DESCRIPTOR_STRIDE (ret, n);
+	  rs = GFC_DESCRIPTOR_STRIDE_BYTES (ret, n);
 	  if (rs != r_ex)
 	    {
 	      do_blocked = false;
 	      break;
 	    }
-	  as = GFC_DESCRIPTOR_STRIDE (array, n);
+	  as = GFC_DESCRIPTOR_STRIDE_BYTES (array, n);
 	  if (as != a_ex)
 	    {
 	      do_blocked = false;
 	      break;
 	    }
+	  index_type extent = GFC_DESCRIPTOR_EXTENT (array, n);
 	  r_ex *= GFC_DESCRIPTOR_EXTENT (ret, n);
-	  a_ex *= GFC_DESCRIPTOR_EXTENT (array, n);
+	  a_ex *= extent;
+	  contiguous_extent = extent;
 	}
     }
   else
@@ -115,9 +119,8 @@ cshift0_i8 (gfc_array_i8 *ret, const gfc_array_i8 *array, ptrdiff_t shift,
       rstride[0] = sizeof (GFC_INTEGER_8);
       roffset = sizeof (GFC_INTEGER_8);
       soffset = sizeof (GFC_INTEGER_8);
-      len = GFC_DESCRIPTOR_STRIDE(array, which)
-	* GFC_DESCRIPTOR_EXTENT(array, which);      
-      shift *= GFC_DESCRIPTOR_STRIDE(array, which);
+      len = contiguous_extent * GFC_DESCRIPTOR_EXTENT(array, which);
+      shift *= contiguous_extent;
       for (dim = which + 1; dim < GFC_DESCRIPTOR_RANK (array); dim++)
 	{
 	  count[n] = 0;
diff --git a/libgfortran/generated/cshift0_r10.c b/libgfortran/generated/cshift0_r10.c
index f40e42627e9e..e4a02140c61f 100644
--- a/libgfortran/generated/cshift0_r10.c
+++ b/libgfortran/generated/cshift0_r10.c
@@ -47,6 +47,7 @@ cshift0_r10 (gfc_array_r10 *ret, const gfc_array_r10 *array, ptrdiff_t shift,
 
   index_type count[GFC_MAX_DIMENSIONS];
   index_type extent[GFC_MAX_DIMENSIONS];
+  index_type contiguous_extent;
   index_type dim;
   index_type len;
   index_type n;
@@ -66,31 +67,34 @@ cshift0_r10 (gfc_array_r10 *ret, const gfc_array_r10 *array, ptrdiff_t shift,
   soffset = 1;
   len = 0;
 
-  r_ex = 1;
-  a_ex = 1;
+  r_ex = sizeof (GFC_REAL_10);
+  a_ex = sizeof (GFC_REAL_10);
 
   if (which > 0)
     {
       /* Test if both ret and array are contiguous.  */
       do_blocked = true;
+      contiguous_extent = 1;
       dim = GFC_DESCRIPTOR_RANK (array);
       for (n = 0; n < dim; n ++)
 	{
 	  index_type rs, as;
-	  rs = GFC_DESCRIPTOR_STRIDE (ret, n);
+	  rs = GFC_DESCRIPTOR_STRIDE_BYTES (ret, n);
 	  if (rs != r_ex)
 	    {
 	      do_blocked = false;
 	      break;
 	    }
-	  as = GFC_DESCRIPTOR_STRIDE (array, n);
+	  as = GFC_DESCRIPTOR_STRIDE_BYTES (array, n);
 	  if (as != a_ex)
 	    {
 	      do_blocked = false;
 	      break;
 	    }
+	  index_type extent = GFC_DESCRIPTOR_EXTENT (array, n);
 	  r_ex *= GFC_DESCRIPTOR_EXTENT (ret, n);
-	  a_ex *= GFC_DESCRIPTOR_EXTENT (array, n);
+	  a_ex *= extent;
+	  contiguous_extent = extent;
 	}
     }
   else
@@ -115,9 +119,8 @@ cshift0_r10 (gfc_array_r10 *ret, const gfc_array_r10 *array, ptrdiff_t shift,
       rstride[0] = sizeof (GFC_REAL_10);
       roffset = sizeof (GFC_REAL_10);
       soffset = sizeof (GFC_REAL_10);
-      len = GFC_DESCRIPTOR_STRIDE(array, which)
-	* GFC_DESCRIPTOR_EXTENT(array, which);      
-      shift *= GFC_DESCRIPTOR_STRIDE(array, which);
+      len = contiguous_extent * GFC_DESCRIPTOR_EXTENT(array, which);
+      shift *= contiguous_extent;
       for (dim = which + 1; dim < GFC_DESCRIPTOR_RANK (array); dim++)
 	{
 	  count[n] = 0;
diff --git a/libgfortran/generated/cshift0_r16.c b/libgfortran/generated/cshift0_r16.c
index cbd98f3e44fd..5dd7678e6835 100644
--- a/libgfortran/generated/cshift0_r16.c
+++ b/libgfortran/generated/cshift0_r16.c
@@ -47,6 +47,7 @@ cshift0_r16 (gfc_array_r16 *ret, const gfc_array_r16 *array, ptrdiff_t shift,
 
   index_type count[GFC_MAX_DIMENSIONS];
   index_type extent[GFC_MAX_DIMENSIONS];
+  index_type contiguous_extent;
   index_type dim;
   index_type len;
   index_type n;
@@ -66,31 +67,34 @@ cshift0_r16 (gfc_array_r16 *ret, const gfc_array_r16 *array, ptrdiff_t shift,
   soffset = 1;
   len = 0;
 
-  r_ex = 1;
-  a_ex = 1;
+  r_ex = sizeof (GFC_REAL_16);
+  a_ex = sizeof (GFC_REAL_16);
 
   if (which > 0)
     {
       /* Test if both ret and array are contiguous.  */
       do_blocked = true;
+      contiguous_extent = 1;
       dim = GFC_DESCRIPTOR_RANK (array);
       for (n = 0; n < dim; n ++)
 	{
 	  index_type rs, as;
-	  rs = GFC_DESCRIPTOR_STRIDE (ret, n);
+	  rs = GFC_DESCRIPTOR_STRIDE_BYTES (ret, n);
 	  if (rs != r_ex)
 	    {
 	      do_blocked = false;
 	      break;
 	    }
-	  as = GFC_DESCRIPTOR_STRIDE (array, n);
+	  as = GFC_DESCRIPTOR_STRIDE_BYTES (array, n);
 	  if (as != a_ex)
 	    {
 	      do_blocked = false;
 	      break;
 	    }
+	  index_type extent = GFC_DESCRIPTOR_EXTENT (array, n);
 	  r_ex *= GFC_DESCRIPTOR_EXTENT (ret, n);
-	  a_ex *= GFC_DESCRIPTOR_EXTENT (array, n);
+	  a_ex *= extent;
+	  contiguous_extent = extent;
 	}
     }
   else
@@ -115,9 +119,8 @@ cshift0_r16 (gfc_array_r16 *ret, const gfc_array_r16 *array, ptrdiff_t shift,
       rstride[0] = sizeof (GFC_REAL_16);
       roffset = sizeof (GFC_REAL_16);
       soffset = sizeof (GFC_REAL_16);
-      len = GFC_DESCRIPTOR_STRIDE(array, which)
-	* GFC_DESCRIPTOR_EXTENT(array, which);      
-      shift *= GFC_DESCRIPTOR_STRIDE(array, which);
+      len = contiguous_extent * GFC_DESCRIPTOR_EXTENT(array, which);
+      shift *= contiguous_extent;
       for (dim = which + 1; dim < GFC_DESCRIPTOR_RANK (array); dim++)
 	{
 	  count[n] = 0;
diff --git a/libgfortran/generated/cshift0_r17.c b/libgfortran/generated/cshift0_r17.c
index a854dffc68c7..7df4915a7eaa 100644
--- a/libgfortran/generated/cshift0_r17.c
+++ b/libgfortran/generated/cshift0_r17.c
@@ -47,6 +47,7 @@ cshift0_r17 (gfc_array_r17 *ret, const gfc_array_r17 *array, ptrdiff_t shift,
 
   index_type count[GFC_MAX_DIMENSIONS];
   index_type extent[GFC_MAX_DIMENSIONS];
+  index_type contiguous_extent;
   index_type dim;
   index_type len;
   index_type n;
@@ -66,31 +67,34 @@ cshift0_r17 (gfc_array_r17 *ret, const gfc_array_r17 *array, ptrdiff_t shift,
   soffset = 1;
   len = 0;
 
-  r_ex = 1;
-  a_ex = 1;
+  r_ex = sizeof (GFC_REAL_17);
+  a_ex = sizeof (GFC_REAL_17);
 
   if (which > 0)
     {
       /* Test if both ret and array are contiguous.  */
       do_blocked = true;
+      contiguous_extent = 1;
       dim = GFC_DESCRIPTOR_RANK (array);
       for (n = 0; n < dim; n ++)
 	{
 	  index_type rs, as;
-	  rs = GFC_DESCRIPTOR_STRIDE (ret, n);
+	  rs = GFC_DESCRIPTOR_STRIDE_BYTES (ret, n);
 	  if (rs != r_ex)
 	    {
 	      do_blocked = false;
 	      break;
 	    }
-	  as = GFC_DESCRIPTOR_STRIDE (array, n);
+	  as = GFC_DESCRIPTOR_STRIDE_BYTES (array, n);
 	  if (as != a_ex)
 	    {
 	      do_blocked = false;
 	      break;
 	    }
+	  index_type extent = GFC_DESCRIPTOR_EXTENT (array, n);
 	  r_ex *= GFC_DESCRIPTOR_EXTENT (ret, n);
-	  a_ex *= GFC_DESCRIPTOR_EXTENT (array, n);
+	  a_ex *= extent;
+	  contiguous_extent = extent;
 	}
     }
   else
@@ -115,9 +119,8 @@ cshift0_r17 (gfc_array_r17 *ret, const gfc_array_r17 *array, ptrdiff_t shift,
       rstride[0] = sizeof (GFC_REAL_17);
       roffset = sizeof (GFC_REAL_17);
       soffset = sizeof (GFC_REAL_17);
-      len = GFC_DESCRIPTOR_STRIDE(array, which)
-	* GFC_DESCRIPTOR_EXTENT(array, which);      
-      shift *= GFC_DESCRIPTOR_STRIDE(array, which);
+      len = contiguous_extent * GFC_DESCRIPTOR_EXTENT(array, which);
+      shift *= contiguous_extent;
       for (dim = which + 1; dim < GFC_DESCRIPTOR_RANK (array); dim++)
 	{
 	  count[n] = 0;
diff --git a/libgfortran/generated/cshift0_r4.c b/libgfortran/generated/cshift0_r4.c
index cb4b93062dad..975bd2057b77 100644
--- a/libgfortran/generated/cshift0_r4.c
+++ b/libgfortran/generated/cshift0_r4.c
@@ -47,6 +47,7 @@ cshift0_r4 (gfc_array_r4 *ret, const gfc_array_r4 *array, ptrdiff_t shift,
 
   index_type count[GFC_MAX_DIMENSIONS];
   index_type extent[GFC_MAX_DIMENSIONS];
+  index_type contiguous_extent;
   index_type dim;
   index_type len;
   index_type n;
@@ -66,31 +67,34 @@ cshift0_r4 (gfc_array_r4 *ret, const gfc_array_r4 *array, ptrdiff_t shift,
   soffset = 1;
   len = 0;
 
-  r_ex = 1;
-  a_ex = 1;
+  r_ex = sizeof (GFC_REAL_4);
+  a_ex = sizeof (GFC_REAL_4);
 
   if (which > 0)
     {
       /* Test if both ret and array are contiguous.  */
       do_blocked = true;
+      contiguous_extent = 1;
       dim = GFC_DESCRIPTOR_RANK (array);
       for (n = 0; n < dim; n ++)
 	{
 	  index_type rs, as;
-	  rs = GFC_DESCRIPTOR_STRIDE (ret, n);
+	  rs = GFC_DESCRIPTOR_STRIDE_BYTES (ret, n);
 	  if (rs != r_ex)
 	    {
 	      do_blocked = false;
 	      break;
 	    }
-	  as = GFC_DESCRIPTOR_STRIDE (array, n);
+	  as = GFC_DESCRIPTOR_STRIDE_BYTES (array, n);
 	  if (as != a_ex)
 	    {
 	      do_blocked = false;
 	      break;
 	    }
+	  index_type extent = GFC_DESCRIPTOR_EXTENT (array, n);
 	  r_ex *= GFC_DESCRIPTOR_EXTENT (ret, n);
-	  a_ex *= GFC_DESCRIPTOR_EXTENT (array, n);
+	  a_ex *= extent;
+	  contiguous_extent = extent;
 	}
     }
   else
@@ -115,9 +119,8 @@ cshift0_r4 (gfc_array_r4 *ret, const gfc_array_r4 *array, ptrdiff_t shift,
       rstride[0] = sizeof (GFC_REAL_4);
       roffset = sizeof (GFC_REAL_4);
       soffset = sizeof (GFC_REAL_4);
-      len = GFC_DESCRIPTOR_STRIDE(array, which)
-	* GFC_DESCRIPTOR_EXTENT(array, which);      
-      shift *= GFC_DESCRIPTOR_STRIDE(array, which);
+      len = contiguous_extent * GFC_DESCRIPTOR_EXTENT(array, which);
+      shift *= contiguous_extent;
       for (dim = which + 1; dim < GFC_DESCRIPTOR_RANK (array); dim++)
 	{
 	  count[n] = 0;
diff --git a/libgfortran/generated/cshift0_r8.c b/libgfortran/generated/cshift0_r8.c
index 8f5a67bffcc6..6503eba9ff79 100644
--- a/libgfortran/generated/cshift0_r8.c
+++ b/libgfortran/generated/cshift0_r8.c
@@ -47,6 +47,7 @@ cshift0_r8 (gfc_array_r8 *ret, const gfc_array_r8 *array, ptrdiff_t shift,
 
   index_type count[GFC_MAX_DIMENSIONS];
   index_type extent[GFC_MAX_DIMENSIONS];
+  index_type contiguous_extent;
   index_type dim;
   index_type len;
   index_type n;
@@ -66,31 +67,34 @@ cshift0_r8 (gfc_array_r8 *ret, const gfc_array_r8 *array, ptrdiff_t shift,
   soffset = 1;
   len = 0;
 
-  r_ex = 1;
-  a_ex = 1;
+  r_ex = sizeof (GFC_REAL_8);
+  a_ex = sizeof (GFC_REAL_8);
 
   if (which > 0)
     {
       /* Test if both ret and array are contiguous.  */
       do_blocked = true;
+      contiguous_extent = 1;
       dim = GFC_DESCRIPTOR_RANK (array);
       for (n = 0; n < dim; n ++)
 	{
 	  index_type rs, as;
-	  rs = GFC_DESCRIPTOR_STRIDE (ret, n);
+	  rs = GFC_DESCRIPTOR_STRIDE_BYTES (ret, n);
 	  if (rs != r_ex)
 	    {
 	      do_blocked = false;
 	      break;
 	    }
-	  as = GFC_DESCRIPTOR_STRIDE (array, n);
+	  as = GFC_DESCRIPTOR_STRIDE_BYTES (array, n);
 	  if (as != a_ex)
 	    {
 	      do_blocked = false;
 	      break;
 	    }
+	  index_type extent = GFC_DESCRIPTOR_EXTENT (array, n);
 	  r_ex *= GFC_DESCRIPTOR_EXTENT (ret, n);
-	  a_ex *= GFC_DESCRIPTOR_EXTENT (array, n);
+	  a_ex *= extent;
+	  contiguous_extent = extent;
 	}
     }
   else
@@ -115,9 +119,8 @@ cshift0_r8 (gfc_array_r8 *ret, const gfc_array_r8 *array, ptrdiff_t shift,
       rstride[0] = sizeof (GFC_REAL_8);
       roffset = sizeof (GFC_REAL_8);
       soffset = sizeof (GFC_REAL_8);
-      len = GFC_DESCRIPTOR_STRIDE(array, which)
-	* GFC_DESCRIPTOR_EXTENT(array, which);      
-      shift *= GFC_DESCRIPTOR_STRIDE(array, which);
+      len = contiguous_extent * GFC_DESCRIPTOR_EXTENT(array, which);
+      shift *= contiguous_extent;
       for (dim = which + 1; dim < GFC_DESCRIPTOR_RANK (array); dim++)
 	{
 	  count[n] = 0;
diff --git a/libgfortran/generated/cshift1_16.c b/libgfortran/generated/cshift1_16.c
index d7fc6012339e..dbd3755ad077 100644
--- a/libgfortran/generated/cshift1_16.c
+++ b/libgfortran/generated/cshift1_16.c
@@ -77,22 +77,20 @@ cshift1 (gfc_array_char * const restrict ret,
 
   if (ret->base_addr == NULL)
     {
+      index_type cnt;
       ret->base_addr = xmallocarray (arraysize, size);
       ret->offset = 0;
       GFC_DTYPE_COPY(ret,array);
+      cnt = 1;
       for (index_type i = 0; i < GFC_DESCRIPTOR_RANK (array); i++)
         {
-	  index_type ub, str;
+	  index_type ub;
 
           ub = GFC_DESCRIPTOR_EXTENT(array,i) - 1;
 
-          if (i == 0)
-            str = 1;
-          else
-	    str = GFC_DESCRIPTOR_EXTENT(ret,i-1) *
-	      GFC_DESCRIPTOR_STRIDE(ret,i-1);
+	  GFC_DESCRIPTOR_DIMENSION_SET(ret, i, 0, ub, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(ret, i, 0, ub, str);
+	  cnt = cnt * GFC_DESCRIPTOR_EXTENT(ret,i);
         }
     }
   else if (unlikely (compile_options.bounds_check))
diff --git a/libgfortran/generated/cshift1_4.c b/libgfortran/generated/cshift1_4.c
index ec92dbc48cda..b1f6e260a331 100644
--- a/libgfortran/generated/cshift1_4.c
+++ b/libgfortran/generated/cshift1_4.c
@@ -77,22 +77,20 @@ cshift1 (gfc_array_char * const restrict ret,
 
   if (ret->base_addr == NULL)
     {
+      index_type cnt;
       ret->base_addr = xmallocarray (arraysize, size);
       ret->offset = 0;
       GFC_DTYPE_COPY(ret,array);
+      cnt = 1;
       for (index_type i = 0; i < GFC_DESCRIPTOR_RANK (array); i++)
         {
-	  index_type ub, str;
+	  index_type ub;
 
           ub = GFC_DESCRIPTOR_EXTENT(array,i) - 1;
 
-          if (i == 0)
-            str = 1;
-          else
-	    str = GFC_DESCRIPTOR_EXTENT(ret,i-1) *
-	      GFC_DESCRIPTOR_STRIDE(ret,i-1);
+	  GFC_DESCRIPTOR_DIMENSION_SET(ret, i, 0, ub, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(ret, i, 0, ub, str);
+	  cnt = cnt * GFC_DESCRIPTOR_EXTENT(ret,i);
         }
     }
   else if (unlikely (compile_options.bounds_check))
diff --git a/libgfortran/generated/cshift1_8.c b/libgfortran/generated/cshift1_8.c
index db8f0df9d19c..1ea7a1946986 100644
--- a/libgfortran/generated/cshift1_8.c
+++ b/libgfortran/generated/cshift1_8.c
@@ -77,22 +77,20 @@ cshift1 (gfc_array_char * const restrict ret,
 
   if (ret->base_addr == NULL)
     {
+      index_type cnt;
       ret->base_addr = xmallocarray (arraysize, size);
       ret->offset = 0;
       GFC_DTYPE_COPY(ret,array);
+      cnt = 1;
       for (index_type i = 0; i < GFC_DESCRIPTOR_RANK (array); i++)
         {
-	  index_type ub, str;
+	  index_type ub;
 
           ub = GFC_DESCRIPTOR_EXTENT(array,i) - 1;
 
-          if (i == 0)
-            str = 1;
-          else
-	    str = GFC_DESCRIPTOR_EXTENT(ret,i-1) *
-	      GFC_DESCRIPTOR_STRIDE(ret,i-1);
+	  GFC_DESCRIPTOR_DIMENSION_SET(ret, i, 0, ub, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(ret, i, 0, ub, str);
+	  cnt = cnt * GFC_DESCRIPTOR_EXTENT(ret,i);
         }
     }
   else if (unlikely (compile_options.bounds_check))
diff --git a/libgfortran/generated/eoshift1_16.c b/libgfortran/generated/eoshift1_16.c
index 6d11687fe9bb..34e9dffdf676 100644
--- a/libgfortran/generated/eoshift1_16.c
+++ b/libgfortran/generated/eoshift1_16.c
@@ -84,21 +84,20 @@ eoshift1 (gfc_array_char * const restrict ret,
   arraysize = size0 ((array_t *) array);
   if (ret->base_addr == NULL)
     {
+      index_type cnt;
+
       ret->offset = 0;
       GFC_DTYPE_COPY(ret,array);
+      cnt = 1;
       for (index_type i = 0; i < GFC_DESCRIPTOR_RANK (array); i++)
         {
-	  index_type ub, str;
+	  index_type ub;
 
 	  ub = GFC_DESCRIPTOR_EXTENT(array,i) - 1;
 
-          if (i == 0)
-            str = 1;
-          else
-            str = GFC_DESCRIPTOR_EXTENT(ret,i-1)
-	      * GFC_DESCRIPTOR_STRIDE(ret,i-1);
+	  GFC_DESCRIPTOR_DIMENSION_SET(ret, i, 0, ub, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(ret, i, 0, ub, str);
+	  cnt = cnt * GFC_DESCRIPTOR_EXTENT(ret,i);
         }
       /* xmallocarray allocates a single byte for zero size.  */
       ret->base_addr = xmallocarray (arraysize, size);
diff --git a/libgfortran/generated/eoshift1_4.c b/libgfortran/generated/eoshift1_4.c
index 9d4e6abeaf1c..1ed3786f2bb5 100644
--- a/libgfortran/generated/eoshift1_4.c
+++ b/libgfortran/generated/eoshift1_4.c
@@ -84,21 +84,20 @@ eoshift1 (gfc_array_char * const restrict ret,
   arraysize = size0 ((array_t *) array);
   if (ret->base_addr == NULL)
     {
+      index_type cnt;
+
       ret->offset = 0;
       GFC_DTYPE_COPY(ret,array);
+      cnt = 1;
       for (index_type i = 0; i < GFC_DESCRIPTOR_RANK (array); i++)
         {
-	  index_type ub, str;
+	  index_type ub;
 
 	  ub = GFC_DESCRIPTOR_EXTENT(array,i) - 1;
 
-          if (i == 0)
-            str = 1;
-          else
-            str = GFC_DESCRIPTOR_EXTENT(ret,i-1)
-	      * GFC_DESCRIPTOR_STRIDE(ret,i-1);
+	  GFC_DESCRIPTOR_DIMENSION_SET(ret, i, 0, ub, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(ret, i, 0, ub, str);
+	  cnt = cnt * GFC_DESCRIPTOR_EXTENT(ret,i);
         }
       /* xmallocarray allocates a single byte for zero size.  */
       ret->base_addr = xmallocarray (arraysize, size);
diff --git a/libgfortran/generated/eoshift1_8.c b/libgfortran/generated/eoshift1_8.c
index dff31fc72add..2f0136fc1236 100644
--- a/libgfortran/generated/eoshift1_8.c
+++ b/libgfortran/generated/eoshift1_8.c
@@ -84,21 +84,20 @@ eoshift1 (gfc_array_char * const restrict ret,
   arraysize = size0 ((array_t *) array);
   if (ret->base_addr == NULL)
     {
+      index_type cnt;
+
       ret->offset = 0;
       GFC_DTYPE_COPY(ret,array);
+      cnt = 1;
       for (index_type i = 0; i < GFC_DESCRIPTOR_RANK (array); i++)
         {
-	  index_type ub, str;
+	  index_type ub;
 
 	  ub = GFC_DESCRIPTOR_EXTENT(array,i) - 1;
 
-          if (i == 0)
-            str = 1;
-          else
-            str = GFC_DESCRIPTOR_EXTENT(ret,i-1)
-	      * GFC_DESCRIPTOR_STRIDE(ret,i-1);
+	  GFC_DESCRIPTOR_DIMENSION_SET(ret, i, 0, ub, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(ret, i, 0, ub, str);
+	  cnt = cnt * GFC_DESCRIPTOR_EXTENT(ret,i);
         }
       /* xmallocarray allocates a single byte for zero size.  */
       ret->base_addr = xmallocarray (arraysize, size);
diff --git a/libgfortran/generated/eoshift3_16.c b/libgfortran/generated/eoshift3_16.c
index 5b8144e65cf0..287110380f96 100644
--- a/libgfortran/generated/eoshift3_16.c
+++ b/libgfortran/generated/eoshift3_16.c
@@ -85,26 +85,23 @@ eoshift3 (gfc_array_char * const restrict ret,
 
   if (ret->base_addr == NULL)
     {
+      index_type cnt;
       ret->base_addr = xmallocarray (arraysize, size);
       ret->offset = 0;
       GFC_DTYPE_COPY(ret,array);
+      cnt = 1;
       for (index_type i = 0; i < GFC_DESCRIPTOR_RANK (array); i++)
         {
-	  index_type ub, str;
+	  index_type ub;
 
 	  ub = GFC_DESCRIPTOR_EXTENT(array,i) - 1;
 
-          if (i == 0)
-            str = 1;
-          else
-            str = GFC_DESCRIPTOR_EXTENT(ret,i-1)
-	      * GFC_DESCRIPTOR_STRIDE(ret,i-1);
+	  GFC_DESCRIPTOR_DIMENSION_SET(ret, i, 0, ub, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(ret, i, 0, ub, str);
+	  cnt = cnt * GFC_DESCRIPTOR_EXTENT(ret,i);
         }
       /* xmallocarray allocates a single byte for zero size.  */
       ret->base_addr = xmallocarray (arraysize, size);
-
     }
   else if (unlikely (compile_options.bounds_check))
     {
diff --git a/libgfortran/generated/eoshift3_4.c b/libgfortran/generated/eoshift3_4.c
index 28d4fa8fb9f0..811288b0657a 100644
--- a/libgfortran/generated/eoshift3_4.c
+++ b/libgfortran/generated/eoshift3_4.c
@@ -85,26 +85,23 @@ eoshift3 (gfc_array_char * const restrict ret,
 
   if (ret->base_addr == NULL)
     {
+      index_type cnt;
       ret->base_addr = xmallocarray (arraysize, size);
       ret->offset = 0;
       GFC_DTYPE_COPY(ret,array);
+      cnt = 1;
       for (index_type i = 0; i < GFC_DESCRIPTOR_RANK (array); i++)
         {
-	  index_type ub, str;
+	  index_type ub;
 
 	  ub = GFC_DESCRIPTOR_EXTENT(array,i) - 1;
 
-          if (i == 0)
-            str = 1;
-          else
-            str = GFC_DESCRIPTOR_EXTENT(ret,i-1)
-	      * GFC_DESCRIPTOR_STRIDE(ret,i-1);
+	  GFC_DESCRIPTOR_DIMENSION_SET(ret, i, 0, ub, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(ret, i, 0, ub, str);
+	  cnt = cnt * GFC_DESCRIPTOR_EXTENT(ret,i);
         }
       /* xmallocarray allocates a single byte for zero size.  */
       ret->base_addr = xmallocarray (arraysize, size);
-
     }
   else if (unlikely (compile_options.bounds_check))
     {
diff --git a/libgfortran/generated/eoshift3_8.c b/libgfortran/generated/eoshift3_8.c
index ec21825295f0..8bd1aeae4e2c 100644
--- a/libgfortran/generated/eoshift3_8.c
+++ b/libgfortran/generated/eoshift3_8.c
@@ -85,26 +85,23 @@ eoshift3 (gfc_array_char * const restrict ret,
 
   if (ret->base_addr == NULL)
     {
+      index_type cnt;
       ret->base_addr = xmallocarray (arraysize, size);
       ret->offset = 0;
       GFC_DTYPE_COPY(ret,array);
+      cnt = 1;
       for (index_type i = 0; i < GFC_DESCRIPTOR_RANK (array); i++)
         {
-	  index_type ub, str;
+	  index_type ub;
 
 	  ub = GFC_DESCRIPTOR_EXTENT(array,i) - 1;
 
-          if (i == 0)
-            str = 1;
-          else
-            str = GFC_DESCRIPTOR_EXTENT(ret,i-1)
-	      * GFC_DESCRIPTOR_STRIDE(ret,i-1);
+	  GFC_DESCRIPTOR_DIMENSION_SET(ret, i, 0, ub, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(ret, i, 0, ub, str);
+	  cnt = cnt * GFC_DESCRIPTOR_EXTENT(ret,i);
         }
       /* xmallocarray allocates a single byte for zero size.  */
       ret->base_addr = xmallocarray (arraysize, size);
-
     }
   else if (unlikely (compile_options.bounds_check))
     {
diff --git a/libgfortran/generated/findloc1_c10.c b/libgfortran/generated/findloc1_c10.c
index cc099b689bd7..3b454ddb42fa 100644
--- a/libgfortran/generated/findloc1_c10.c
+++ b/libgfortran/generated/findloc1_c10.c
@@ -85,25 +85,21 @@ findloc1_c10 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
@@ -270,25 +266,21 @@ mfindloc1_c10 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
@@ -443,25 +435,21 @@ sfindloc1_c10 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/findloc1_c16.c b/libgfortran/generated/findloc1_c16.c
index 83cee88cb95d..9748767a09ab 100644
--- a/libgfortran/generated/findloc1_c16.c
+++ b/libgfortran/generated/findloc1_c16.c
@@ -85,25 +85,21 @@ findloc1_c16 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
@@ -270,25 +266,21 @@ mfindloc1_c16 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
@@ -443,25 +435,21 @@ sfindloc1_c16 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/findloc1_c17.c b/libgfortran/generated/findloc1_c17.c
index c6392edcce34..8609d0cf0531 100644
--- a/libgfortran/generated/findloc1_c17.c
+++ b/libgfortran/generated/findloc1_c17.c
@@ -85,25 +85,21 @@ findloc1_c17 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
@@ -270,25 +266,21 @@ mfindloc1_c17 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
@@ -443,25 +435,21 @@ sfindloc1_c17 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/findloc1_c4.c b/libgfortran/generated/findloc1_c4.c
index be8551fc29aa..998faeeeb306 100644
--- a/libgfortran/generated/findloc1_c4.c
+++ b/libgfortran/generated/findloc1_c4.c
@@ -85,25 +85,21 @@ findloc1_c4 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
@@ -270,25 +266,21 @@ mfindloc1_c4 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
@@ -443,25 +435,21 @@ sfindloc1_c4 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/findloc1_c8.c b/libgfortran/generated/findloc1_c8.c
index 1bbef53695fa..0d05b4bd313f 100644
--- a/libgfortran/generated/findloc1_c8.c
+++ b/libgfortran/generated/findloc1_c8.c
@@ -85,25 +85,21 @@ findloc1_c8 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
@@ -270,25 +266,21 @@ mfindloc1_c8 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
@@ -443,25 +435,21 @@ sfindloc1_c8 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/findloc1_i1.c b/libgfortran/generated/findloc1_i1.c
index 7be40974e8a5..7cd6bf1999b3 100644
--- a/libgfortran/generated/findloc1_i1.c
+++ b/libgfortran/generated/findloc1_i1.c
@@ -85,25 +85,21 @@ findloc1_i1 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
@@ -270,25 +266,21 @@ mfindloc1_i1 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
@@ -443,25 +435,21 @@ sfindloc1_i1 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/findloc1_i16.c b/libgfortran/generated/findloc1_i16.c
index 909a618201fa..3565b1b05733 100644
--- a/libgfortran/generated/findloc1_i16.c
+++ b/libgfortran/generated/findloc1_i16.c
@@ -85,25 +85,21 @@ findloc1_i16 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
@@ -270,25 +266,21 @@ mfindloc1_i16 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
@@ -443,25 +435,21 @@ sfindloc1_i16 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/findloc1_i2.c b/libgfortran/generated/findloc1_i2.c
index 05f7604bf5c8..58961f901dfc 100644
--- a/libgfortran/generated/findloc1_i2.c
+++ b/libgfortran/generated/findloc1_i2.c
@@ -85,25 +85,21 @@ findloc1_i2 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
@@ -270,25 +266,21 @@ mfindloc1_i2 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
@@ -443,25 +435,21 @@ sfindloc1_i2 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/findloc1_i4.c b/libgfortran/generated/findloc1_i4.c
index 07954ffc581d..c4b490e9d4c3 100644
--- a/libgfortran/generated/findloc1_i4.c
+++ b/libgfortran/generated/findloc1_i4.c
@@ -85,25 +85,21 @@ findloc1_i4 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
@@ -270,25 +266,21 @@ mfindloc1_i4 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
@@ -443,25 +435,21 @@ sfindloc1_i4 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/findloc1_i8.c b/libgfortran/generated/findloc1_i8.c
index ed3f62b99f75..b4206973e6e4 100644
--- a/libgfortran/generated/findloc1_i8.c
+++ b/libgfortran/generated/findloc1_i8.c
@@ -85,25 +85,21 @@ findloc1_i8 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
@@ -270,25 +266,21 @@ mfindloc1_i8 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
@@ -443,25 +435,21 @@ sfindloc1_i8 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/findloc1_r10.c b/libgfortran/generated/findloc1_r10.c
index 738268678301..af77051ffc6a 100644
--- a/libgfortran/generated/findloc1_r10.c
+++ b/libgfortran/generated/findloc1_r10.c
@@ -85,25 +85,21 @@ findloc1_r10 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
@@ -270,25 +266,21 @@ mfindloc1_r10 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
@@ -443,25 +435,21 @@ sfindloc1_r10 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/findloc1_r16.c b/libgfortran/generated/findloc1_r16.c
index 7f9fa5ecfc3f..03356a36d3b8 100644
--- a/libgfortran/generated/findloc1_r16.c
+++ b/libgfortran/generated/findloc1_r16.c
@@ -85,25 +85,21 @@ findloc1_r16 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
@@ -270,25 +266,21 @@ mfindloc1_r16 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
@@ -443,25 +435,21 @@ sfindloc1_r16 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/findloc1_r17.c b/libgfortran/generated/findloc1_r17.c
index 95a79ac93aff..89576d07882c 100644
--- a/libgfortran/generated/findloc1_r17.c
+++ b/libgfortran/generated/findloc1_r17.c
@@ -85,25 +85,21 @@ findloc1_r17 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
@@ -270,25 +266,21 @@ mfindloc1_r17 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
@@ -443,25 +435,21 @@ sfindloc1_r17 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/findloc1_r4.c b/libgfortran/generated/findloc1_r4.c
index f9d3f59341c9..d2cbd296632e 100644
--- a/libgfortran/generated/findloc1_r4.c
+++ b/libgfortran/generated/findloc1_r4.c
@@ -85,25 +85,21 @@ findloc1_r4 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
@@ -270,25 +266,21 @@ mfindloc1_r4 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
@@ -443,25 +435,21 @@ sfindloc1_r4 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/findloc1_r8.c b/libgfortran/generated/findloc1_r8.c
index 83ad7544d5f0..21f42e727da0 100644
--- a/libgfortran/generated/findloc1_r8.c
+++ b/libgfortran/generated/findloc1_r8.c
@@ -85,25 +85,21 @@ findloc1_r8 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
@@ -270,25 +266,21 @@ mfindloc1_r8 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
@@ -443,25 +435,21 @@ sfindloc1_r8 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/findloc1_s1.c b/libgfortran/generated/findloc1_s1.c
index a243e4606ac6..766c4b4108dc 100644
--- a/libgfortran/generated/findloc1_s1.c
+++ b/libgfortran/generated/findloc1_s1.c
@@ -87,25 +87,21 @@ findloc1_s1 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
@@ -272,25 +268,21 @@ mfindloc1_s1 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
@@ -445,25 +437,21 @@ sfindloc1_s1 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/findloc1_s4.c b/libgfortran/generated/findloc1_s4.c
index 4214658bd007..ee005a67f12b 100644
--- a/libgfortran/generated/findloc1_s4.c
+++ b/libgfortran/generated/findloc1_s4.c
@@ -87,25 +87,21 @@ findloc1_s4 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
@@ -272,25 +268,21 @@ mfindloc1_s4 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
@@ -445,25 +437,21 @@ sfindloc1_s4 (gfc_array_index_type * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (index_type));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (index_type));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/iall_i1.c b/libgfortran/generated/iall_i1.c
index 42d9f85db170..aac7375e350c 100644
--- a/libgfortran/generated/iall_i1.c
+++ b/libgfortran/generated/iall_i1.c
@@ -86,16 +86,14 @@ iall_i1 (gfc_array_i1 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
@@ -103,10 +101,8 @@ iall_i1 (gfc_array_i1 * const restrict retarray,
       retarray->dtype.elem_len = sizeof (GFC_INTEGER_1);
       retarray->span = sizeof (GFC_INTEGER_1);
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_1));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_1));
+      if (cnt == 0)
 	return;
     }
   else
@@ -280,27 +276,23 @@ miall_i1 (gfc_array_i1 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str= GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
       retarray->offset = 0;
       retarray->dtype.rank = rank;
       retarray->dtype.elem_len = sizeof (GFC_INTEGER_1);
       retarray->span = sizeof (GFC_INTEGER_1);
 
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_1));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_1));
+      if (cnt == 0)
 	return;
     }
   else
@@ -442,25 +434,21 @@ siall_i1 (gfc_array_i1 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_1));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_1));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/iall_i16.c b/libgfortran/generated/iall_i16.c
index 84e8cf758f70..3c63d805f6fd 100644
--- a/libgfortran/generated/iall_i16.c
+++ b/libgfortran/generated/iall_i16.c
@@ -86,16 +86,14 @@ iall_i16 (gfc_array_i16 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
@@ -103,10 +101,8 @@ iall_i16 (gfc_array_i16 * const restrict retarray,
       retarray->dtype.elem_len = sizeof (GFC_INTEGER_16);
       retarray->span = sizeof (GFC_INTEGER_16);
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_16));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_16));
+      if (cnt == 0)
 	return;
     }
   else
@@ -280,27 +276,23 @@ miall_i16 (gfc_array_i16 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str= GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
       retarray->offset = 0;
       retarray->dtype.rank = rank;
       retarray->dtype.elem_len = sizeof (GFC_INTEGER_16);
       retarray->span = sizeof (GFC_INTEGER_16);
 
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_16));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_16));
+      if (cnt == 0)
 	return;
     }
   else
@@ -442,25 +434,21 @@ siall_i16 (gfc_array_i16 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_16));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_16));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/iall_i2.c b/libgfortran/generated/iall_i2.c
index b315017b24c4..a7177235f5cd 100644
--- a/libgfortran/generated/iall_i2.c
+++ b/libgfortran/generated/iall_i2.c
@@ -86,16 +86,14 @@ iall_i2 (gfc_array_i2 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
@@ -103,10 +101,8 @@ iall_i2 (gfc_array_i2 * const restrict retarray,
       retarray->dtype.elem_len = sizeof (GFC_INTEGER_2);
       retarray->span = sizeof (GFC_INTEGER_2);
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_2));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_2));
+      if (cnt == 0)
 	return;
     }
   else
@@ -280,27 +276,23 @@ miall_i2 (gfc_array_i2 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str= GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
       retarray->offset = 0;
       retarray->dtype.rank = rank;
       retarray->dtype.elem_len = sizeof (GFC_INTEGER_2);
       retarray->span = sizeof (GFC_INTEGER_2);
 
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_2));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_2));
+      if (cnt == 0)
 	return;
     }
   else
@@ -442,25 +434,21 @@ siall_i2 (gfc_array_i2 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_2));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_2));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/iall_i4.c b/libgfortran/generated/iall_i4.c
index 09dbbe620d40..9a2ead05787d 100644
--- a/libgfortran/generated/iall_i4.c
+++ b/libgfortran/generated/iall_i4.c
@@ -86,16 +86,14 @@ iall_i4 (gfc_array_i4 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
@@ -103,10 +101,8 @@ iall_i4 (gfc_array_i4 * const restrict retarray,
       retarray->dtype.elem_len = sizeof (GFC_INTEGER_4);
       retarray->span = sizeof (GFC_INTEGER_4);
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_4));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_4));
+      if (cnt == 0)
 	return;
     }
   else
@@ -280,27 +276,23 @@ miall_i4 (gfc_array_i4 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str= GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
       retarray->offset = 0;
       retarray->dtype.rank = rank;
       retarray->dtype.elem_len = sizeof (GFC_INTEGER_4);
       retarray->span = sizeof (GFC_INTEGER_4);
 
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_4));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_4));
+      if (cnt == 0)
 	return;
     }
   else
@@ -442,25 +434,21 @@ siall_i4 (gfc_array_i4 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_4));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_4));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/iall_i8.c b/libgfortran/generated/iall_i8.c
index 1d27dd0d6afa..b02c1bd6adf8 100644
--- a/libgfortran/generated/iall_i8.c
+++ b/libgfortran/generated/iall_i8.c
@@ -86,16 +86,14 @@ iall_i8 (gfc_array_i8 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
@@ -103,10 +101,8 @@ iall_i8 (gfc_array_i8 * const restrict retarray,
       retarray->dtype.elem_len = sizeof (GFC_INTEGER_8);
       retarray->span = sizeof (GFC_INTEGER_8);
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_8));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_8));
+      if (cnt == 0)
 	return;
     }
   else
@@ -280,27 +276,23 @@ miall_i8 (gfc_array_i8 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str= GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
       retarray->offset = 0;
       retarray->dtype.rank = rank;
       retarray->dtype.elem_len = sizeof (GFC_INTEGER_8);
       retarray->span = sizeof (GFC_INTEGER_8);
 
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_8));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_8));
+      if (cnt == 0)
 	return;
     }
   else
@@ -442,25 +434,21 @@ siall_i8 (gfc_array_i8 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_8));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_8));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/iany_i1.c b/libgfortran/generated/iany_i1.c
index fc9b6c8e8b84..2ca7cd5e21c7 100644
--- a/libgfortran/generated/iany_i1.c
+++ b/libgfortran/generated/iany_i1.c
@@ -86,16 +86,14 @@ iany_i1 (gfc_array_i1 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
@@ -103,10 +101,8 @@ iany_i1 (gfc_array_i1 * const restrict retarray,
       retarray->dtype.elem_len = sizeof (GFC_INTEGER_1);
       retarray->span = sizeof (GFC_INTEGER_1);
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_1));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_1));
+      if (cnt == 0)
 	return;
     }
   else
@@ -280,27 +276,23 @@ miany_i1 (gfc_array_i1 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str= GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
       retarray->offset = 0;
       retarray->dtype.rank = rank;
       retarray->dtype.elem_len = sizeof (GFC_INTEGER_1);
       retarray->span = sizeof (GFC_INTEGER_1);
 
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_1));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_1));
+      if (cnt == 0)
 	return;
     }
   else
@@ -442,25 +434,21 @@ siany_i1 (gfc_array_i1 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_1));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_1));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/iany_i16.c b/libgfortran/generated/iany_i16.c
index bad7ba7f3ff6..051f898687e7 100644
--- a/libgfortran/generated/iany_i16.c
+++ b/libgfortran/generated/iany_i16.c
@@ -86,16 +86,14 @@ iany_i16 (gfc_array_i16 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
@@ -103,10 +101,8 @@ iany_i16 (gfc_array_i16 * const restrict retarray,
       retarray->dtype.elem_len = sizeof (GFC_INTEGER_16);
       retarray->span = sizeof (GFC_INTEGER_16);
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_16));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_16));
+      if (cnt == 0)
 	return;
     }
   else
@@ -280,27 +276,23 @@ miany_i16 (gfc_array_i16 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str= GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
       retarray->offset = 0;
       retarray->dtype.rank = rank;
       retarray->dtype.elem_len = sizeof (GFC_INTEGER_16);
       retarray->span = sizeof (GFC_INTEGER_16);
 
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_16));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_16));
+      if (cnt == 0)
 	return;
     }
   else
@@ -442,25 +434,21 @@ siany_i16 (gfc_array_i16 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_16));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_16));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/iany_i2.c b/libgfortran/generated/iany_i2.c
index 9f02eb5be0ba..dc644d92311f 100644
--- a/libgfortran/generated/iany_i2.c
+++ b/libgfortran/generated/iany_i2.c
@@ -86,16 +86,14 @@ iany_i2 (gfc_array_i2 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
@@ -103,10 +101,8 @@ iany_i2 (gfc_array_i2 * const restrict retarray,
       retarray->dtype.elem_len = sizeof (GFC_INTEGER_2);
       retarray->span = sizeof (GFC_INTEGER_2);
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_2));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_2));
+      if (cnt == 0)
 	return;
     }
   else
@@ -280,27 +276,23 @@ miany_i2 (gfc_array_i2 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str= GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
       retarray->offset = 0;
       retarray->dtype.rank = rank;
       retarray->dtype.elem_len = sizeof (GFC_INTEGER_2);
       retarray->span = sizeof (GFC_INTEGER_2);
 
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_2));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_2));
+      if (cnt == 0)
 	return;
     }
   else
@@ -442,25 +434,21 @@ siany_i2 (gfc_array_i2 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_2));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_2));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/iany_i4.c b/libgfortran/generated/iany_i4.c
index 48a6f6f66417..654d299ce8bb 100644
--- a/libgfortran/generated/iany_i4.c
+++ b/libgfortran/generated/iany_i4.c
@@ -86,16 +86,14 @@ iany_i4 (gfc_array_i4 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
@@ -103,10 +101,8 @@ iany_i4 (gfc_array_i4 * const restrict retarray,
       retarray->dtype.elem_len = sizeof (GFC_INTEGER_4);
       retarray->span = sizeof (GFC_INTEGER_4);
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_4));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_4));
+      if (cnt == 0)
 	return;
     }
   else
@@ -280,27 +276,23 @@ miany_i4 (gfc_array_i4 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str= GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
       retarray->offset = 0;
       retarray->dtype.rank = rank;
       retarray->dtype.elem_len = sizeof (GFC_INTEGER_4);
       retarray->span = sizeof (GFC_INTEGER_4);
 
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_4));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_4));
+      if (cnt == 0)
 	return;
     }
   else
@@ -442,25 +434,21 @@ siany_i4 (gfc_array_i4 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_4));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_4));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/iany_i8.c b/libgfortran/generated/iany_i8.c
index 2fff60d3b2af..9f9ac72423b6 100644
--- a/libgfortran/generated/iany_i8.c
+++ b/libgfortran/generated/iany_i8.c
@@ -86,16 +86,14 @@ iany_i8 (gfc_array_i8 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
@@ -103,10 +101,8 @@ iany_i8 (gfc_array_i8 * const restrict retarray,
       retarray->dtype.elem_len = sizeof (GFC_INTEGER_8);
       retarray->span = sizeof (GFC_INTEGER_8);
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_8));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_8));
+      if (cnt == 0)
 	return;
     }
   else
@@ -280,27 +276,23 @@ miany_i8 (gfc_array_i8 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str= GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
       retarray->offset = 0;
       retarray->dtype.rank = rank;
       retarray->dtype.elem_len = sizeof (GFC_INTEGER_8);
       retarray->span = sizeof (GFC_INTEGER_8);
 
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_8));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_8));
+      if (cnt == 0)
 	return;
     }
   else
@@ -442,25 +434,21 @@ siany_i8 (gfc_array_i8 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_8));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_8));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/iparity_i1.c b/libgfortran/generated/iparity_i1.c
index 7128083ecc3a..d9aa623b65b0 100644
--- a/libgfortran/generated/iparity_i1.c
+++ b/libgfortran/generated/iparity_i1.c
@@ -86,16 +86,14 @@ iparity_i1 (gfc_array_i1 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
@@ -103,10 +101,8 @@ iparity_i1 (gfc_array_i1 * const restrict retarray,
       retarray->dtype.elem_len = sizeof (GFC_INTEGER_1);
       retarray->span = sizeof (GFC_INTEGER_1);
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_1));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_1));
+      if (cnt == 0)
 	return;
     }
   else
@@ -280,27 +276,23 @@ miparity_i1 (gfc_array_i1 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str= GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
       retarray->offset = 0;
       retarray->dtype.rank = rank;
       retarray->dtype.elem_len = sizeof (GFC_INTEGER_1);
       retarray->span = sizeof (GFC_INTEGER_1);
 
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_1));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_1));
+      if (cnt == 0)
 	return;
     }
   else
@@ -442,25 +434,21 @@ siparity_i1 (gfc_array_i1 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_1));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_1));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/iparity_i16.c b/libgfortran/generated/iparity_i16.c
index 5f16ac265a15..8f3d00104fe4 100644
--- a/libgfortran/generated/iparity_i16.c
+++ b/libgfortran/generated/iparity_i16.c
@@ -86,16 +86,14 @@ iparity_i16 (gfc_array_i16 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
@@ -103,10 +101,8 @@ iparity_i16 (gfc_array_i16 * const restrict retarray,
       retarray->dtype.elem_len = sizeof (GFC_INTEGER_16);
       retarray->span = sizeof (GFC_INTEGER_16);
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_16));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_16));
+      if (cnt == 0)
 	return;
     }
   else
@@ -280,27 +276,23 @@ miparity_i16 (gfc_array_i16 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str= GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
       retarray->offset = 0;
       retarray->dtype.rank = rank;
       retarray->dtype.elem_len = sizeof (GFC_INTEGER_16);
       retarray->span = sizeof (GFC_INTEGER_16);
 
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_16));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_16));
+      if (cnt == 0)
 	return;
     }
   else
@@ -442,25 +434,21 @@ siparity_i16 (gfc_array_i16 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_16));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_16));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/iparity_i2.c b/libgfortran/generated/iparity_i2.c
index 5c9781a6cd53..2b4dd6108fdc 100644
--- a/libgfortran/generated/iparity_i2.c
+++ b/libgfortran/generated/iparity_i2.c
@@ -86,16 +86,14 @@ iparity_i2 (gfc_array_i2 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
@@ -103,10 +101,8 @@ iparity_i2 (gfc_array_i2 * const restrict retarray,
       retarray->dtype.elem_len = sizeof (GFC_INTEGER_2);
       retarray->span = sizeof (GFC_INTEGER_2);
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_2));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_2));
+      if (cnt == 0)
 	return;
     }
   else
@@ -280,27 +276,23 @@ miparity_i2 (gfc_array_i2 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str= GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
       retarray->offset = 0;
       retarray->dtype.rank = rank;
       retarray->dtype.elem_len = sizeof (GFC_INTEGER_2);
       retarray->span = sizeof (GFC_INTEGER_2);
 
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_2));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_2));
+      if (cnt == 0)
 	return;
     }
   else
@@ -442,25 +434,21 @@ siparity_i2 (gfc_array_i2 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_2));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_2));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/iparity_i4.c b/libgfortran/generated/iparity_i4.c
index 5af120d748da..b9a0d8a0db0e 100644
--- a/libgfortran/generated/iparity_i4.c
+++ b/libgfortran/generated/iparity_i4.c
@@ -86,16 +86,14 @@ iparity_i4 (gfc_array_i4 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
@@ -103,10 +101,8 @@ iparity_i4 (gfc_array_i4 * const restrict retarray,
       retarray->dtype.elem_len = sizeof (GFC_INTEGER_4);
       retarray->span = sizeof (GFC_INTEGER_4);
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_4));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_4));
+      if (cnt == 0)
 	return;
     }
   else
@@ -280,27 +276,23 @@ miparity_i4 (gfc_array_i4 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str= GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
       retarray->offset = 0;
       retarray->dtype.rank = rank;
       retarray->dtype.elem_len = sizeof (GFC_INTEGER_4);
       retarray->span = sizeof (GFC_INTEGER_4);
 
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_4));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_4));
+      if (cnt == 0)
 	return;
     }
   else
@@ -442,25 +434,21 @@ siparity_i4 (gfc_array_i4 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_4));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_4));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/iparity_i8.c b/libgfortran/generated/iparity_i8.c
index 9673ea4ed10a..fb73eff0044c 100644
--- a/libgfortran/generated/iparity_i8.c
+++ b/libgfortran/generated/iparity_i8.c
@@ -86,16 +86,14 @@ iparity_i8 (gfc_array_i8 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
@@ -103,10 +101,8 @@ iparity_i8 (gfc_array_i8 * const restrict retarray,
       retarray->dtype.elem_len = sizeof (GFC_INTEGER_8);
       retarray->span = sizeof (GFC_INTEGER_8);
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_8));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_8));
+      if (cnt == 0)
 	return;
     }
   else
@@ -280,27 +276,23 @@ miparity_i8 (gfc_array_i8 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str= GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
       retarray->offset = 0;
       retarray->dtype.rank = rank;
       retarray->dtype.elem_len = sizeof (GFC_INTEGER_8);
       retarray->span = sizeof (GFC_INTEGER_8);
 
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_8));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_8));
+      if (cnt == 0)
 	return;
     }
   else
@@ -442,25 +434,21 @@ siparity_i8 (gfc_array_i8 * const restrict retarray,
 
   if (retarray->base_addr == NULL)
     {
-      size_t alloc_size, str;
+      size_t cnt;
 
+      cnt = 1;
       for (n = 0; n < rank; n++)
 	{
-	  if (n == 0)
-	    str = 1;
-	  else
-	    str = GFC_DESCRIPTOR_STRIDE(retarray,n-1) * extent[n-1];
+	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, cnt);
 
-	  GFC_DESCRIPTOR_DIMENSION_SET(retarray, n, 0, extent[n] - 1, str);
+	  cnt = cnt * extent[n];
 	}
 
       retarray->offset = 0;
       retarray->dtype.rank = rank;
 
-      alloc_size = GFC_DESCRIPTOR_STRIDE(retarray,rank-1) * extent[rank-1];
-
-      retarray->base_addr = xmallocarray (alloc_size, sizeof (GFC_INTEGER_8));
-      if (alloc_size == 0)
+      retarray->base_addr = xmallocarray (cnt, sizeof (GFC_INTEGER_8));
+      if (cnt == 0)
 	return;
     }
   else
diff --git a/libgfortran/generated/matmul_c10.c b/libgfortran/generated/matmul_c10.c
index f5c298aa0c4f..a5b9493b3f61 100644
--- a/libgfortran/generated/matmul_c10.c
+++ b/libgfortran/generated/matmul_c10.c
@@ -92,7 +92,7 @@ matmul_c10_avx (gfc_array_c10 * const restrict retarray,
   const GFC_COMPLEX_10 * restrict bbase;
   GFC_COMPLEX_10 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -183,12 +183,11 @@ matmul_c10_avx (gfc_array_c10 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -257,15 +256,19 @@ matmul_c10_avx (gfc_array_c10 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_COMPLEX_10)
+      && (axstride_bytes == sizeof (GFC_COMPLEX_10)
+	  || aystride_bytes == sizeof (GFC_COMPLEX_10))
+      && (bxstride_bytes == sizeof (GFC_COMPLEX_10)
+	  || bystride_bytes == sizeof (GFC_COMPLEX_10))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_COMPLEX_10 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_COMPLEX_10)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_COMPLEX_10)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -274,12 +277,12 @@ matmul_c10_avx (gfc_array_c10 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_COMPLEX_10) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_COMPLEX_10) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -288,7 +291,9 @@ matmul_c10_avx (gfc_array_c10 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_COMPLEX_10)
+      && axstride_bytes == sizeof (GFC_COMPLEX_10)
+      && bxstride_bytes == sizeof (GFC_COMPLEX_10)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -341,7 +346,7 @@ matmul_c10_avx (gfc_array_c10 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_COMPLEX_10))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -567,7 +572,9 @@ matmul_c10_avx (gfc_array_c10 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_COMPLEX_10)
+	   && aystride_bytes == sizeof (GFC_COMPLEX_10)
+	   && bxstride_bytes == sizeof (GFC_COMPLEX_10))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -620,7 +627,7 @@ matmul_c10_avx (gfc_array_c10 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -677,7 +684,7 @@ matmul_c10_avx2 (gfc_array_c10 * const restrict retarray,
   const GFC_COMPLEX_10 * restrict bbase;
   GFC_COMPLEX_10 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -768,12 +775,11 @@ matmul_c10_avx2 (gfc_array_c10 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -842,15 +848,19 @@ matmul_c10_avx2 (gfc_array_c10 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_COMPLEX_10)
+      && (axstride_bytes == sizeof (GFC_COMPLEX_10)
+	  || aystride_bytes == sizeof (GFC_COMPLEX_10))
+      && (bxstride_bytes == sizeof (GFC_COMPLEX_10)
+	  || bystride_bytes == sizeof (GFC_COMPLEX_10))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_COMPLEX_10 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_COMPLEX_10)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_COMPLEX_10)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -859,12 +869,12 @@ matmul_c10_avx2 (gfc_array_c10 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_COMPLEX_10) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_COMPLEX_10) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -873,7 +883,9 @@ matmul_c10_avx2 (gfc_array_c10 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_COMPLEX_10)
+      && axstride_bytes == sizeof (GFC_COMPLEX_10)
+      && bxstride_bytes == sizeof (GFC_COMPLEX_10)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -926,7 +938,7 @@ matmul_c10_avx2 (gfc_array_c10 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_COMPLEX_10))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -1152,7 +1164,9 @@ matmul_c10_avx2 (gfc_array_c10 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_COMPLEX_10)
+	   && aystride_bytes == sizeof (GFC_COMPLEX_10)
+	   && bxstride_bytes == sizeof (GFC_COMPLEX_10))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -1205,7 +1219,7 @@ matmul_c10_avx2 (gfc_array_c10 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -1262,7 +1276,7 @@ matmul_c10_avx512f (gfc_array_c10 * const restrict retarray,
   const GFC_COMPLEX_10 * restrict bbase;
   GFC_COMPLEX_10 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -1353,12 +1367,11 @@ matmul_c10_avx512f (gfc_array_c10 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -1427,15 +1440,19 @@ matmul_c10_avx512f (gfc_array_c10 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_COMPLEX_10)
+      && (axstride_bytes == sizeof (GFC_COMPLEX_10)
+	  || aystride_bytes == sizeof (GFC_COMPLEX_10))
+      && (bxstride_bytes == sizeof (GFC_COMPLEX_10)
+	  || bystride_bytes == sizeof (GFC_COMPLEX_10))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_COMPLEX_10 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_COMPLEX_10)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_COMPLEX_10)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -1444,12 +1461,12 @@ matmul_c10_avx512f (gfc_array_c10 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_COMPLEX_10) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_COMPLEX_10) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -1458,7 +1475,9 @@ matmul_c10_avx512f (gfc_array_c10 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_COMPLEX_10)
+      && axstride_bytes == sizeof (GFC_COMPLEX_10)
+      && bxstride_bytes == sizeof (GFC_COMPLEX_10)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -1511,7 +1530,7 @@ matmul_c10_avx512f (gfc_array_c10 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_COMPLEX_10))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -1737,7 +1756,9 @@ matmul_c10_avx512f (gfc_array_c10 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_COMPLEX_10)
+	   && aystride_bytes == sizeof (GFC_COMPLEX_10)
+	   && bxstride_bytes == sizeof (GFC_COMPLEX_10))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -1790,7 +1811,7 @@ matmul_c10_avx512f (gfc_array_c10 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -1861,7 +1882,7 @@ matmul_c10_vanilla (gfc_array_c10 * const restrict retarray,
   const GFC_COMPLEX_10 * restrict bbase;
   GFC_COMPLEX_10 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -1952,12 +1973,11 @@ matmul_c10_vanilla (gfc_array_c10 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -2026,15 +2046,19 @@ matmul_c10_vanilla (gfc_array_c10 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_COMPLEX_10)
+      && (axstride_bytes == sizeof (GFC_COMPLEX_10)
+	  || aystride_bytes == sizeof (GFC_COMPLEX_10))
+      && (bxstride_bytes == sizeof (GFC_COMPLEX_10)
+	  || bystride_bytes == sizeof (GFC_COMPLEX_10))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_COMPLEX_10 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_COMPLEX_10)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_COMPLEX_10)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -2043,12 +2067,12 @@ matmul_c10_vanilla (gfc_array_c10 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_COMPLEX_10) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_COMPLEX_10) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -2057,7 +2081,9 @@ matmul_c10_vanilla (gfc_array_c10 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_COMPLEX_10)
+      && axstride_bytes == sizeof (GFC_COMPLEX_10)
+      && bxstride_bytes == sizeof (GFC_COMPLEX_10)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -2110,7 +2136,7 @@ matmul_c10_vanilla (gfc_array_c10 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_COMPLEX_10))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -2336,7 +2362,9 @@ matmul_c10_vanilla (gfc_array_c10 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_COMPLEX_10)
+	   && aystride_bytes == sizeof (GFC_COMPLEX_10)
+	   && bxstride_bytes == sizeof (GFC_COMPLEX_10))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -2389,7 +2417,7 @@ matmul_c10_vanilla (gfc_array_c10 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -2519,7 +2547,7 @@ matmul_c10 (gfc_array_c10 * const restrict retarray,
   const GFC_COMPLEX_10 * restrict bbase;
   GFC_COMPLEX_10 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -2610,12 +2638,11 @@ matmul_c10 (gfc_array_c10 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -2684,15 +2711,19 @@ matmul_c10 (gfc_array_c10 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_COMPLEX_10)
+      && (axstride_bytes == sizeof (GFC_COMPLEX_10)
+	  || aystride_bytes == sizeof (GFC_COMPLEX_10))
+      && (bxstride_bytes == sizeof (GFC_COMPLEX_10)
+	  || bystride_bytes == sizeof (GFC_COMPLEX_10))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_COMPLEX_10 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_COMPLEX_10)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_COMPLEX_10)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -2701,12 +2732,12 @@ matmul_c10 (gfc_array_c10 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_COMPLEX_10) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_COMPLEX_10) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -2715,7 +2746,9 @@ matmul_c10 (gfc_array_c10 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_COMPLEX_10)
+      && axstride_bytes == sizeof (GFC_COMPLEX_10)
+      && bxstride_bytes == sizeof (GFC_COMPLEX_10)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -2768,7 +2801,7 @@ matmul_c10 (gfc_array_c10 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_COMPLEX_10))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -2994,7 +3027,9 @@ matmul_c10 (gfc_array_c10 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_COMPLEX_10)
+	   && aystride_bytes == sizeof (GFC_COMPLEX_10)
+	   && bxstride_bytes == sizeof (GFC_COMPLEX_10))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -3047,7 +3082,7 @@ matmul_c10 (gfc_array_c10 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
diff --git a/libgfortran/generated/matmul_c16.c b/libgfortran/generated/matmul_c16.c
index 8d592540b553..bbe80e3ba462 100644
--- a/libgfortran/generated/matmul_c16.c
+++ b/libgfortran/generated/matmul_c16.c
@@ -92,7 +92,7 @@ matmul_c16_avx (gfc_array_c16 * const restrict retarray,
   const GFC_COMPLEX_16 * restrict bbase;
   GFC_COMPLEX_16 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -183,12 +183,11 @@ matmul_c16_avx (gfc_array_c16 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -257,15 +256,19 @@ matmul_c16_avx (gfc_array_c16 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_COMPLEX_16)
+      && (axstride_bytes == sizeof (GFC_COMPLEX_16)
+	  || aystride_bytes == sizeof (GFC_COMPLEX_16))
+      && (bxstride_bytes == sizeof (GFC_COMPLEX_16)
+	  || bystride_bytes == sizeof (GFC_COMPLEX_16))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_COMPLEX_16 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_COMPLEX_16)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_COMPLEX_16)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -274,12 +277,12 @@ matmul_c16_avx (gfc_array_c16 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_COMPLEX_16) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_COMPLEX_16) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -288,7 +291,9 @@ matmul_c16_avx (gfc_array_c16 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_COMPLEX_16)
+      && axstride_bytes == sizeof (GFC_COMPLEX_16)
+      && bxstride_bytes == sizeof (GFC_COMPLEX_16)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -341,7 +346,7 @@ matmul_c16_avx (gfc_array_c16 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_COMPLEX_16))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -567,7 +572,9 @@ matmul_c16_avx (gfc_array_c16 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_COMPLEX_16)
+	   && aystride_bytes == sizeof (GFC_COMPLEX_16)
+	   && bxstride_bytes == sizeof (GFC_COMPLEX_16))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -620,7 +627,7 @@ matmul_c16_avx (gfc_array_c16 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -677,7 +684,7 @@ matmul_c16_avx2 (gfc_array_c16 * const restrict retarray,
   const GFC_COMPLEX_16 * restrict bbase;
   GFC_COMPLEX_16 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -768,12 +775,11 @@ matmul_c16_avx2 (gfc_array_c16 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -842,15 +848,19 @@ matmul_c16_avx2 (gfc_array_c16 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_COMPLEX_16)
+      && (axstride_bytes == sizeof (GFC_COMPLEX_16)
+	  || aystride_bytes == sizeof (GFC_COMPLEX_16))
+      && (bxstride_bytes == sizeof (GFC_COMPLEX_16)
+	  || bystride_bytes == sizeof (GFC_COMPLEX_16))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_COMPLEX_16 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_COMPLEX_16)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_COMPLEX_16)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -859,12 +869,12 @@ matmul_c16_avx2 (gfc_array_c16 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_COMPLEX_16) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_COMPLEX_16) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -873,7 +883,9 @@ matmul_c16_avx2 (gfc_array_c16 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_COMPLEX_16)
+      && axstride_bytes == sizeof (GFC_COMPLEX_16)
+      && bxstride_bytes == sizeof (GFC_COMPLEX_16)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -926,7 +938,7 @@ matmul_c16_avx2 (gfc_array_c16 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_COMPLEX_16))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -1152,7 +1164,9 @@ matmul_c16_avx2 (gfc_array_c16 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_COMPLEX_16)
+	   && aystride_bytes == sizeof (GFC_COMPLEX_16)
+	   && bxstride_bytes == sizeof (GFC_COMPLEX_16))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -1205,7 +1219,7 @@ matmul_c16_avx2 (gfc_array_c16 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -1262,7 +1276,7 @@ matmul_c16_avx512f (gfc_array_c16 * const restrict retarray,
   const GFC_COMPLEX_16 * restrict bbase;
   GFC_COMPLEX_16 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -1353,12 +1367,11 @@ matmul_c16_avx512f (gfc_array_c16 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -1427,15 +1440,19 @@ matmul_c16_avx512f (gfc_array_c16 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_COMPLEX_16)
+      && (axstride_bytes == sizeof (GFC_COMPLEX_16)
+	  || aystride_bytes == sizeof (GFC_COMPLEX_16))
+      && (bxstride_bytes == sizeof (GFC_COMPLEX_16)
+	  || bystride_bytes == sizeof (GFC_COMPLEX_16))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_COMPLEX_16 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_COMPLEX_16)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_COMPLEX_16)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -1444,12 +1461,12 @@ matmul_c16_avx512f (gfc_array_c16 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_COMPLEX_16) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_COMPLEX_16) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -1458,7 +1475,9 @@ matmul_c16_avx512f (gfc_array_c16 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_COMPLEX_16)
+      && axstride_bytes == sizeof (GFC_COMPLEX_16)
+      && bxstride_bytes == sizeof (GFC_COMPLEX_16)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -1511,7 +1530,7 @@ matmul_c16_avx512f (gfc_array_c16 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_COMPLEX_16))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -1737,7 +1756,9 @@ matmul_c16_avx512f (gfc_array_c16 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_COMPLEX_16)
+	   && aystride_bytes == sizeof (GFC_COMPLEX_16)
+	   && bxstride_bytes == sizeof (GFC_COMPLEX_16))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -1790,7 +1811,7 @@ matmul_c16_avx512f (gfc_array_c16 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -1861,7 +1882,7 @@ matmul_c16_vanilla (gfc_array_c16 * const restrict retarray,
   const GFC_COMPLEX_16 * restrict bbase;
   GFC_COMPLEX_16 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -1952,12 +1973,11 @@ matmul_c16_vanilla (gfc_array_c16 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -2026,15 +2046,19 @@ matmul_c16_vanilla (gfc_array_c16 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_COMPLEX_16)
+      && (axstride_bytes == sizeof (GFC_COMPLEX_16)
+	  || aystride_bytes == sizeof (GFC_COMPLEX_16))
+      && (bxstride_bytes == sizeof (GFC_COMPLEX_16)
+	  || bystride_bytes == sizeof (GFC_COMPLEX_16))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_COMPLEX_16 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_COMPLEX_16)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_COMPLEX_16)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -2043,12 +2067,12 @@ matmul_c16_vanilla (gfc_array_c16 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_COMPLEX_16) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_COMPLEX_16) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -2057,7 +2081,9 @@ matmul_c16_vanilla (gfc_array_c16 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_COMPLEX_16)
+      && axstride_bytes == sizeof (GFC_COMPLEX_16)
+      && bxstride_bytes == sizeof (GFC_COMPLEX_16)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -2110,7 +2136,7 @@ matmul_c16_vanilla (gfc_array_c16 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_COMPLEX_16))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -2336,7 +2362,9 @@ matmul_c16_vanilla (gfc_array_c16 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_COMPLEX_16)
+	   && aystride_bytes == sizeof (GFC_COMPLEX_16)
+	   && bxstride_bytes == sizeof (GFC_COMPLEX_16))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -2389,7 +2417,7 @@ matmul_c16_vanilla (gfc_array_c16 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -2519,7 +2547,7 @@ matmul_c16 (gfc_array_c16 * const restrict retarray,
   const GFC_COMPLEX_16 * restrict bbase;
   GFC_COMPLEX_16 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -2610,12 +2638,11 @@ matmul_c16 (gfc_array_c16 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -2684,15 +2711,19 @@ matmul_c16 (gfc_array_c16 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_COMPLEX_16)
+      && (axstride_bytes == sizeof (GFC_COMPLEX_16)
+	  || aystride_bytes == sizeof (GFC_COMPLEX_16))
+      && (bxstride_bytes == sizeof (GFC_COMPLEX_16)
+	  || bystride_bytes == sizeof (GFC_COMPLEX_16))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_COMPLEX_16 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_COMPLEX_16)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_COMPLEX_16)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -2701,12 +2732,12 @@ matmul_c16 (gfc_array_c16 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_COMPLEX_16) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_COMPLEX_16) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -2715,7 +2746,9 @@ matmul_c16 (gfc_array_c16 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_COMPLEX_16)
+      && axstride_bytes == sizeof (GFC_COMPLEX_16)
+      && bxstride_bytes == sizeof (GFC_COMPLEX_16)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -2768,7 +2801,7 @@ matmul_c16 (gfc_array_c16 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_COMPLEX_16))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -2994,7 +3027,9 @@ matmul_c16 (gfc_array_c16 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_COMPLEX_16)
+	   && aystride_bytes == sizeof (GFC_COMPLEX_16)
+	   && bxstride_bytes == sizeof (GFC_COMPLEX_16))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -3047,7 +3082,7 @@ matmul_c16 (gfc_array_c16 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
diff --git a/libgfortran/generated/matmul_c17.c b/libgfortran/generated/matmul_c17.c
index 47b76385fb67..dd83147774f4 100644
--- a/libgfortran/generated/matmul_c17.c
+++ b/libgfortran/generated/matmul_c17.c
@@ -92,7 +92,7 @@ matmul_c17_avx (gfc_array_c17 * const restrict retarray,
   const GFC_COMPLEX_17 * restrict bbase;
   GFC_COMPLEX_17 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -183,12 +183,11 @@ matmul_c17_avx (gfc_array_c17 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -257,15 +256,19 @@ matmul_c17_avx (gfc_array_c17 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_COMPLEX_17)
+      && (axstride_bytes == sizeof (GFC_COMPLEX_17)
+	  || aystride_bytes == sizeof (GFC_COMPLEX_17))
+      && (bxstride_bytes == sizeof (GFC_COMPLEX_17)
+	  || bystride_bytes == sizeof (GFC_COMPLEX_17))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_COMPLEX_17 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_COMPLEX_17)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_COMPLEX_17)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -274,12 +277,12 @@ matmul_c17_avx (gfc_array_c17 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_COMPLEX_17) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_COMPLEX_17) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -288,7 +291,9 @@ matmul_c17_avx (gfc_array_c17 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_COMPLEX_17)
+      && axstride_bytes == sizeof (GFC_COMPLEX_17)
+      && bxstride_bytes == sizeof (GFC_COMPLEX_17)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -341,7 +346,7 @@ matmul_c17_avx (gfc_array_c17 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_COMPLEX_17))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -567,7 +572,9 @@ matmul_c17_avx (gfc_array_c17 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_COMPLEX_17)
+	   && aystride_bytes == sizeof (GFC_COMPLEX_17)
+	   && bxstride_bytes == sizeof (GFC_COMPLEX_17))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -620,7 +627,7 @@ matmul_c17_avx (gfc_array_c17 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -677,7 +684,7 @@ matmul_c17_avx2 (gfc_array_c17 * const restrict retarray,
   const GFC_COMPLEX_17 * restrict bbase;
   GFC_COMPLEX_17 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -768,12 +775,11 @@ matmul_c17_avx2 (gfc_array_c17 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -842,15 +848,19 @@ matmul_c17_avx2 (gfc_array_c17 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_COMPLEX_17)
+      && (axstride_bytes == sizeof (GFC_COMPLEX_17)
+	  || aystride_bytes == sizeof (GFC_COMPLEX_17))
+      && (bxstride_bytes == sizeof (GFC_COMPLEX_17)
+	  || bystride_bytes == sizeof (GFC_COMPLEX_17))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_COMPLEX_17 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_COMPLEX_17)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_COMPLEX_17)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -859,12 +869,12 @@ matmul_c17_avx2 (gfc_array_c17 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_COMPLEX_17) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_COMPLEX_17) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -873,7 +883,9 @@ matmul_c17_avx2 (gfc_array_c17 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_COMPLEX_17)
+      && axstride_bytes == sizeof (GFC_COMPLEX_17)
+      && bxstride_bytes == sizeof (GFC_COMPLEX_17)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -926,7 +938,7 @@ matmul_c17_avx2 (gfc_array_c17 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_COMPLEX_17))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -1152,7 +1164,9 @@ matmul_c17_avx2 (gfc_array_c17 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_COMPLEX_17)
+	   && aystride_bytes == sizeof (GFC_COMPLEX_17)
+	   && bxstride_bytes == sizeof (GFC_COMPLEX_17))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -1205,7 +1219,7 @@ matmul_c17_avx2 (gfc_array_c17 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -1262,7 +1276,7 @@ matmul_c17_avx512f (gfc_array_c17 * const restrict retarray,
   const GFC_COMPLEX_17 * restrict bbase;
   GFC_COMPLEX_17 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -1353,12 +1367,11 @@ matmul_c17_avx512f (gfc_array_c17 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -1427,15 +1440,19 @@ matmul_c17_avx512f (gfc_array_c17 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_COMPLEX_17)
+      && (axstride_bytes == sizeof (GFC_COMPLEX_17)
+	  || aystride_bytes == sizeof (GFC_COMPLEX_17))
+      && (bxstride_bytes == sizeof (GFC_COMPLEX_17)
+	  || bystride_bytes == sizeof (GFC_COMPLEX_17))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_COMPLEX_17 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_COMPLEX_17)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_COMPLEX_17)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -1444,12 +1461,12 @@ matmul_c17_avx512f (gfc_array_c17 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_COMPLEX_17) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_COMPLEX_17) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -1458,7 +1475,9 @@ matmul_c17_avx512f (gfc_array_c17 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_COMPLEX_17)
+      && axstride_bytes == sizeof (GFC_COMPLEX_17)
+      && bxstride_bytes == sizeof (GFC_COMPLEX_17)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -1511,7 +1530,7 @@ matmul_c17_avx512f (gfc_array_c17 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_COMPLEX_17))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -1737,7 +1756,9 @@ matmul_c17_avx512f (gfc_array_c17 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_COMPLEX_17)
+	   && aystride_bytes == sizeof (GFC_COMPLEX_17)
+	   && bxstride_bytes == sizeof (GFC_COMPLEX_17))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -1790,7 +1811,7 @@ matmul_c17_avx512f (gfc_array_c17 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -1861,7 +1882,7 @@ matmul_c17_vanilla (gfc_array_c17 * const restrict retarray,
   const GFC_COMPLEX_17 * restrict bbase;
   GFC_COMPLEX_17 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -1952,12 +1973,11 @@ matmul_c17_vanilla (gfc_array_c17 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -2026,15 +2046,19 @@ matmul_c17_vanilla (gfc_array_c17 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_COMPLEX_17)
+      && (axstride_bytes == sizeof (GFC_COMPLEX_17)
+	  || aystride_bytes == sizeof (GFC_COMPLEX_17))
+      && (bxstride_bytes == sizeof (GFC_COMPLEX_17)
+	  || bystride_bytes == sizeof (GFC_COMPLEX_17))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_COMPLEX_17 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_COMPLEX_17)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_COMPLEX_17)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -2043,12 +2067,12 @@ matmul_c17_vanilla (gfc_array_c17 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_COMPLEX_17) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_COMPLEX_17) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -2057,7 +2081,9 @@ matmul_c17_vanilla (gfc_array_c17 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_COMPLEX_17)
+      && axstride_bytes == sizeof (GFC_COMPLEX_17)
+      && bxstride_bytes == sizeof (GFC_COMPLEX_17)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -2110,7 +2136,7 @@ matmul_c17_vanilla (gfc_array_c17 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_COMPLEX_17))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -2336,7 +2362,9 @@ matmul_c17_vanilla (gfc_array_c17 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_COMPLEX_17)
+	   && aystride_bytes == sizeof (GFC_COMPLEX_17)
+	   && bxstride_bytes == sizeof (GFC_COMPLEX_17))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -2389,7 +2417,7 @@ matmul_c17_vanilla (gfc_array_c17 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -2519,7 +2547,7 @@ matmul_c17 (gfc_array_c17 * const restrict retarray,
   const GFC_COMPLEX_17 * restrict bbase;
   GFC_COMPLEX_17 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -2610,12 +2638,11 @@ matmul_c17 (gfc_array_c17 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -2684,15 +2711,19 @@ matmul_c17 (gfc_array_c17 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_COMPLEX_17)
+      && (axstride_bytes == sizeof (GFC_COMPLEX_17)
+	  || aystride_bytes == sizeof (GFC_COMPLEX_17))
+      && (bxstride_bytes == sizeof (GFC_COMPLEX_17)
+	  || bystride_bytes == sizeof (GFC_COMPLEX_17))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_COMPLEX_17 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_COMPLEX_17)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_COMPLEX_17)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -2701,12 +2732,12 @@ matmul_c17 (gfc_array_c17 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_COMPLEX_17) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_COMPLEX_17) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -2715,7 +2746,9 @@ matmul_c17 (gfc_array_c17 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_COMPLEX_17)
+      && axstride_bytes == sizeof (GFC_COMPLEX_17)
+      && bxstride_bytes == sizeof (GFC_COMPLEX_17)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -2768,7 +2801,7 @@ matmul_c17 (gfc_array_c17 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_COMPLEX_17))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -2994,7 +3027,9 @@ matmul_c17 (gfc_array_c17 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_COMPLEX_17)
+	   && aystride_bytes == sizeof (GFC_COMPLEX_17)
+	   && bxstride_bytes == sizeof (GFC_COMPLEX_17))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -3047,7 +3082,7 @@ matmul_c17 (gfc_array_c17 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
diff --git a/libgfortran/generated/matmul_c4.c b/libgfortran/generated/matmul_c4.c
index afb9f4f1473d..eecad1f8e0af 100644
--- a/libgfortran/generated/matmul_c4.c
+++ b/libgfortran/generated/matmul_c4.c
@@ -92,7 +92,7 @@ matmul_c4_avx (gfc_array_c4 * const restrict retarray,
   const GFC_COMPLEX_4 * restrict bbase;
   GFC_COMPLEX_4 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -183,12 +183,11 @@ matmul_c4_avx (gfc_array_c4 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -257,15 +256,19 @@ matmul_c4_avx (gfc_array_c4 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_COMPLEX_4)
+      && (axstride_bytes == sizeof (GFC_COMPLEX_4)
+	  || aystride_bytes == sizeof (GFC_COMPLEX_4))
+      && (bxstride_bytes == sizeof (GFC_COMPLEX_4)
+	  || bystride_bytes == sizeof (GFC_COMPLEX_4))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_COMPLEX_4 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_COMPLEX_4)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_COMPLEX_4)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -274,12 +277,12 @@ matmul_c4_avx (gfc_array_c4 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_COMPLEX_4) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_COMPLEX_4) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -288,7 +291,9 @@ matmul_c4_avx (gfc_array_c4 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_COMPLEX_4)
+      && axstride_bytes == sizeof (GFC_COMPLEX_4)
+      && bxstride_bytes == sizeof (GFC_COMPLEX_4)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -341,7 +346,7 @@ matmul_c4_avx (gfc_array_c4 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_COMPLEX_4))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -567,7 +572,9 @@ matmul_c4_avx (gfc_array_c4 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_COMPLEX_4)
+	   && aystride_bytes == sizeof (GFC_COMPLEX_4)
+	   && bxstride_bytes == sizeof (GFC_COMPLEX_4))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -620,7 +627,7 @@ matmul_c4_avx (gfc_array_c4 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -677,7 +684,7 @@ matmul_c4_avx2 (gfc_array_c4 * const restrict retarray,
   const GFC_COMPLEX_4 * restrict bbase;
   GFC_COMPLEX_4 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -768,12 +775,11 @@ matmul_c4_avx2 (gfc_array_c4 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -842,15 +848,19 @@ matmul_c4_avx2 (gfc_array_c4 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_COMPLEX_4)
+      && (axstride_bytes == sizeof (GFC_COMPLEX_4)
+	  || aystride_bytes == sizeof (GFC_COMPLEX_4))
+      && (bxstride_bytes == sizeof (GFC_COMPLEX_4)
+	  || bystride_bytes == sizeof (GFC_COMPLEX_4))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_COMPLEX_4 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_COMPLEX_4)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_COMPLEX_4)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -859,12 +869,12 @@ matmul_c4_avx2 (gfc_array_c4 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_COMPLEX_4) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_COMPLEX_4) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -873,7 +883,9 @@ matmul_c4_avx2 (gfc_array_c4 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_COMPLEX_4)
+      && axstride_bytes == sizeof (GFC_COMPLEX_4)
+      && bxstride_bytes == sizeof (GFC_COMPLEX_4)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -926,7 +938,7 @@ matmul_c4_avx2 (gfc_array_c4 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_COMPLEX_4))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -1152,7 +1164,9 @@ matmul_c4_avx2 (gfc_array_c4 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_COMPLEX_4)
+	   && aystride_bytes == sizeof (GFC_COMPLEX_4)
+	   && bxstride_bytes == sizeof (GFC_COMPLEX_4))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -1205,7 +1219,7 @@ matmul_c4_avx2 (gfc_array_c4 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -1262,7 +1276,7 @@ matmul_c4_avx512f (gfc_array_c4 * const restrict retarray,
   const GFC_COMPLEX_4 * restrict bbase;
   GFC_COMPLEX_4 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -1353,12 +1367,11 @@ matmul_c4_avx512f (gfc_array_c4 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -1427,15 +1440,19 @@ matmul_c4_avx512f (gfc_array_c4 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_COMPLEX_4)
+      && (axstride_bytes == sizeof (GFC_COMPLEX_4)
+	  || aystride_bytes == sizeof (GFC_COMPLEX_4))
+      && (bxstride_bytes == sizeof (GFC_COMPLEX_4)
+	  || bystride_bytes == sizeof (GFC_COMPLEX_4))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_COMPLEX_4 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_COMPLEX_4)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_COMPLEX_4)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -1444,12 +1461,12 @@ matmul_c4_avx512f (gfc_array_c4 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_COMPLEX_4) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_COMPLEX_4) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -1458,7 +1475,9 @@ matmul_c4_avx512f (gfc_array_c4 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_COMPLEX_4)
+      && axstride_bytes == sizeof (GFC_COMPLEX_4)
+      && bxstride_bytes == sizeof (GFC_COMPLEX_4)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -1511,7 +1530,7 @@ matmul_c4_avx512f (gfc_array_c4 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_COMPLEX_4))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -1737,7 +1756,9 @@ matmul_c4_avx512f (gfc_array_c4 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_COMPLEX_4)
+	   && aystride_bytes == sizeof (GFC_COMPLEX_4)
+	   && bxstride_bytes == sizeof (GFC_COMPLEX_4))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -1790,7 +1811,7 @@ matmul_c4_avx512f (gfc_array_c4 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -1861,7 +1882,7 @@ matmul_c4_vanilla (gfc_array_c4 * const restrict retarray,
   const GFC_COMPLEX_4 * restrict bbase;
   GFC_COMPLEX_4 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -1952,12 +1973,11 @@ matmul_c4_vanilla (gfc_array_c4 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -2026,15 +2046,19 @@ matmul_c4_vanilla (gfc_array_c4 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_COMPLEX_4)
+      && (axstride_bytes == sizeof (GFC_COMPLEX_4)
+	  || aystride_bytes == sizeof (GFC_COMPLEX_4))
+      && (bxstride_bytes == sizeof (GFC_COMPLEX_4)
+	  || bystride_bytes == sizeof (GFC_COMPLEX_4))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_COMPLEX_4 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_COMPLEX_4)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_COMPLEX_4)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -2043,12 +2067,12 @@ matmul_c4_vanilla (gfc_array_c4 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_COMPLEX_4) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_COMPLEX_4) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -2057,7 +2081,9 @@ matmul_c4_vanilla (gfc_array_c4 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_COMPLEX_4)
+      && axstride_bytes == sizeof (GFC_COMPLEX_4)
+      && bxstride_bytes == sizeof (GFC_COMPLEX_4)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -2110,7 +2136,7 @@ matmul_c4_vanilla (gfc_array_c4 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_COMPLEX_4))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -2336,7 +2362,9 @@ matmul_c4_vanilla (gfc_array_c4 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_COMPLEX_4)
+	   && aystride_bytes == sizeof (GFC_COMPLEX_4)
+	   && bxstride_bytes == sizeof (GFC_COMPLEX_4))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -2389,7 +2417,7 @@ matmul_c4_vanilla (gfc_array_c4 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -2519,7 +2547,7 @@ matmul_c4 (gfc_array_c4 * const restrict retarray,
   const GFC_COMPLEX_4 * restrict bbase;
   GFC_COMPLEX_4 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -2610,12 +2638,11 @@ matmul_c4 (gfc_array_c4 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -2684,15 +2711,19 @@ matmul_c4 (gfc_array_c4 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_COMPLEX_4)
+      && (axstride_bytes == sizeof (GFC_COMPLEX_4)
+	  || aystride_bytes == sizeof (GFC_COMPLEX_4))
+      && (bxstride_bytes == sizeof (GFC_COMPLEX_4)
+	  || bystride_bytes == sizeof (GFC_COMPLEX_4))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_COMPLEX_4 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_COMPLEX_4)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_COMPLEX_4)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -2701,12 +2732,12 @@ matmul_c4 (gfc_array_c4 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_COMPLEX_4) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_COMPLEX_4) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -2715,7 +2746,9 @@ matmul_c4 (gfc_array_c4 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_COMPLEX_4)
+      && axstride_bytes == sizeof (GFC_COMPLEX_4)
+      && bxstride_bytes == sizeof (GFC_COMPLEX_4)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -2768,7 +2801,7 @@ matmul_c4 (gfc_array_c4 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_COMPLEX_4))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -2994,7 +3027,9 @@ matmul_c4 (gfc_array_c4 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_COMPLEX_4)
+	   && aystride_bytes == sizeof (GFC_COMPLEX_4)
+	   && bxstride_bytes == sizeof (GFC_COMPLEX_4))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -3047,7 +3082,7 @@ matmul_c4 (gfc_array_c4 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
diff --git a/libgfortran/generated/matmul_c8.c b/libgfortran/generated/matmul_c8.c
index bc37144bfbbd..71ced811798c 100644
--- a/libgfortran/generated/matmul_c8.c
+++ b/libgfortran/generated/matmul_c8.c
@@ -92,7 +92,7 @@ matmul_c8_avx (gfc_array_c8 * const restrict retarray,
   const GFC_COMPLEX_8 * restrict bbase;
   GFC_COMPLEX_8 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -183,12 +183,11 @@ matmul_c8_avx (gfc_array_c8 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -257,15 +256,19 @@ matmul_c8_avx (gfc_array_c8 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_COMPLEX_8)
+      && (axstride_bytes == sizeof (GFC_COMPLEX_8)
+	  || aystride_bytes == sizeof (GFC_COMPLEX_8))
+      && (bxstride_bytes == sizeof (GFC_COMPLEX_8)
+	  || bystride_bytes == sizeof (GFC_COMPLEX_8))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_COMPLEX_8 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_COMPLEX_8)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_COMPLEX_8)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -274,12 +277,12 @@ matmul_c8_avx (gfc_array_c8 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_COMPLEX_8) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_COMPLEX_8) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -288,7 +291,9 @@ matmul_c8_avx (gfc_array_c8 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_COMPLEX_8)
+      && axstride_bytes == sizeof (GFC_COMPLEX_8)
+      && bxstride_bytes == sizeof (GFC_COMPLEX_8)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -341,7 +346,7 @@ matmul_c8_avx (gfc_array_c8 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_COMPLEX_8))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -567,7 +572,9 @@ matmul_c8_avx (gfc_array_c8 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_COMPLEX_8)
+	   && aystride_bytes == sizeof (GFC_COMPLEX_8)
+	   && bxstride_bytes == sizeof (GFC_COMPLEX_8))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -620,7 +627,7 @@ matmul_c8_avx (gfc_array_c8 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -677,7 +684,7 @@ matmul_c8_avx2 (gfc_array_c8 * const restrict retarray,
   const GFC_COMPLEX_8 * restrict bbase;
   GFC_COMPLEX_8 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -768,12 +775,11 @@ matmul_c8_avx2 (gfc_array_c8 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -842,15 +848,19 @@ matmul_c8_avx2 (gfc_array_c8 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_COMPLEX_8)
+      && (axstride_bytes == sizeof (GFC_COMPLEX_8)
+	  || aystride_bytes == sizeof (GFC_COMPLEX_8))
+      && (bxstride_bytes == sizeof (GFC_COMPLEX_8)
+	  || bystride_bytes == sizeof (GFC_COMPLEX_8))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_COMPLEX_8 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_COMPLEX_8)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_COMPLEX_8)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -859,12 +869,12 @@ matmul_c8_avx2 (gfc_array_c8 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_COMPLEX_8) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_COMPLEX_8) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -873,7 +883,9 @@ matmul_c8_avx2 (gfc_array_c8 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_COMPLEX_8)
+      && axstride_bytes == sizeof (GFC_COMPLEX_8)
+      && bxstride_bytes == sizeof (GFC_COMPLEX_8)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -926,7 +938,7 @@ matmul_c8_avx2 (gfc_array_c8 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_COMPLEX_8))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -1152,7 +1164,9 @@ matmul_c8_avx2 (gfc_array_c8 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_COMPLEX_8)
+	   && aystride_bytes == sizeof (GFC_COMPLEX_8)
+	   && bxstride_bytes == sizeof (GFC_COMPLEX_8))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -1205,7 +1219,7 @@ matmul_c8_avx2 (gfc_array_c8 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -1262,7 +1276,7 @@ matmul_c8_avx512f (gfc_array_c8 * const restrict retarray,
   const GFC_COMPLEX_8 * restrict bbase;
   GFC_COMPLEX_8 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -1353,12 +1367,11 @@ matmul_c8_avx512f (gfc_array_c8 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -1427,15 +1440,19 @@ matmul_c8_avx512f (gfc_array_c8 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_COMPLEX_8)
+      && (axstride_bytes == sizeof (GFC_COMPLEX_8)
+	  || aystride_bytes == sizeof (GFC_COMPLEX_8))
+      && (bxstride_bytes == sizeof (GFC_COMPLEX_8)
+	  || bystride_bytes == sizeof (GFC_COMPLEX_8))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_COMPLEX_8 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_COMPLEX_8)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_COMPLEX_8)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -1444,12 +1461,12 @@ matmul_c8_avx512f (gfc_array_c8 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_COMPLEX_8) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_COMPLEX_8) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -1458,7 +1475,9 @@ matmul_c8_avx512f (gfc_array_c8 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_COMPLEX_8)
+      && axstride_bytes == sizeof (GFC_COMPLEX_8)
+      && bxstride_bytes == sizeof (GFC_COMPLEX_8)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -1511,7 +1530,7 @@ matmul_c8_avx512f (gfc_array_c8 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_COMPLEX_8))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -1737,7 +1756,9 @@ matmul_c8_avx512f (gfc_array_c8 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_COMPLEX_8)
+	   && aystride_bytes == sizeof (GFC_COMPLEX_8)
+	   && bxstride_bytes == sizeof (GFC_COMPLEX_8))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -1790,7 +1811,7 @@ matmul_c8_avx512f (gfc_array_c8 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -1861,7 +1882,7 @@ matmul_c8_vanilla (gfc_array_c8 * const restrict retarray,
   const GFC_COMPLEX_8 * restrict bbase;
   GFC_COMPLEX_8 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -1952,12 +1973,11 @@ matmul_c8_vanilla (gfc_array_c8 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -2026,15 +2046,19 @@ matmul_c8_vanilla (gfc_array_c8 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_COMPLEX_8)
+      && (axstride_bytes == sizeof (GFC_COMPLEX_8)
+	  || aystride_bytes == sizeof (GFC_COMPLEX_8))
+      && (bxstride_bytes == sizeof (GFC_COMPLEX_8)
+	  || bystride_bytes == sizeof (GFC_COMPLEX_8))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_COMPLEX_8 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_COMPLEX_8)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_COMPLEX_8)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -2043,12 +2067,12 @@ matmul_c8_vanilla (gfc_array_c8 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_COMPLEX_8) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_COMPLEX_8) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -2057,7 +2081,9 @@ matmul_c8_vanilla (gfc_array_c8 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_COMPLEX_8)
+      && axstride_bytes == sizeof (GFC_COMPLEX_8)
+      && bxstride_bytes == sizeof (GFC_COMPLEX_8)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -2110,7 +2136,7 @@ matmul_c8_vanilla (gfc_array_c8 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_COMPLEX_8))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -2336,7 +2362,9 @@ matmul_c8_vanilla (gfc_array_c8 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_COMPLEX_8)
+	   && aystride_bytes == sizeof (GFC_COMPLEX_8)
+	   && bxstride_bytes == sizeof (GFC_COMPLEX_8))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -2389,7 +2417,7 @@ matmul_c8_vanilla (gfc_array_c8 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -2519,7 +2547,7 @@ matmul_c8 (gfc_array_c8 * const restrict retarray,
   const GFC_COMPLEX_8 * restrict bbase;
   GFC_COMPLEX_8 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -2610,12 +2638,11 @@ matmul_c8 (gfc_array_c8 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -2684,15 +2711,19 @@ matmul_c8 (gfc_array_c8 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_COMPLEX_8)
+      && (axstride_bytes == sizeof (GFC_COMPLEX_8)
+	  || aystride_bytes == sizeof (GFC_COMPLEX_8))
+      && (bxstride_bytes == sizeof (GFC_COMPLEX_8)
+	  || bystride_bytes == sizeof (GFC_COMPLEX_8))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_COMPLEX_8 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_COMPLEX_8)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_COMPLEX_8)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -2701,12 +2732,12 @@ matmul_c8 (gfc_array_c8 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_COMPLEX_8) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_COMPLEX_8) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -2715,7 +2746,9 @@ matmul_c8 (gfc_array_c8 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_COMPLEX_8)
+      && axstride_bytes == sizeof (GFC_COMPLEX_8)
+      && bxstride_bytes == sizeof (GFC_COMPLEX_8)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -2768,7 +2801,7 @@ matmul_c8 (gfc_array_c8 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_COMPLEX_8))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -2994,7 +3027,9 @@ matmul_c8 (gfc_array_c8 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_COMPLEX_8)
+	   && aystride_bytes == sizeof (GFC_COMPLEX_8)
+	   && bxstride_bytes == sizeof (GFC_COMPLEX_8))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -3047,7 +3082,7 @@ matmul_c8 (gfc_array_c8 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
diff --git a/libgfortran/generated/matmul_i1.c b/libgfortran/generated/matmul_i1.c
index 2b45d14eeb19..ea47cda8257e 100644
--- a/libgfortran/generated/matmul_i1.c
+++ b/libgfortran/generated/matmul_i1.c
@@ -92,7 +92,7 @@ matmul_i1_avx (gfc_array_m1 * const restrict retarray,
   const GFC_UINTEGER_1 * restrict bbase;
   GFC_UINTEGER_1 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -183,12 +183,11 @@ matmul_i1_avx (gfc_array_m1 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -257,15 +256,19 @@ matmul_i1_avx (gfc_array_m1 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_UINTEGER_1)
+      && (axstride_bytes == sizeof (GFC_UINTEGER_1)
+	  || aystride_bytes == sizeof (GFC_UINTEGER_1))
+      && (bxstride_bytes == sizeof (GFC_UINTEGER_1)
+	  || bystride_bytes == sizeof (GFC_UINTEGER_1))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_UINTEGER_1 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_UINTEGER_1)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_UINTEGER_1)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -274,12 +277,12 @@ matmul_i1_avx (gfc_array_m1 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_UINTEGER_1) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_UINTEGER_1) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -288,7 +291,9 @@ matmul_i1_avx (gfc_array_m1 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_UINTEGER_1)
+      && axstride_bytes == sizeof (GFC_UINTEGER_1)
+      && bxstride_bytes == sizeof (GFC_UINTEGER_1)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -341,7 +346,7 @@ matmul_i1_avx (gfc_array_m1 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_UINTEGER_1))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -567,7 +572,9 @@ matmul_i1_avx (gfc_array_m1 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_UINTEGER_1)
+	   && aystride_bytes == sizeof (GFC_UINTEGER_1)
+	   && bxstride_bytes == sizeof (GFC_UINTEGER_1))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -620,7 +627,7 @@ matmul_i1_avx (gfc_array_m1 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -677,7 +684,7 @@ matmul_i1_avx2 (gfc_array_m1 * const restrict retarray,
   const GFC_UINTEGER_1 * restrict bbase;
   GFC_UINTEGER_1 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -768,12 +775,11 @@ matmul_i1_avx2 (gfc_array_m1 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -842,15 +848,19 @@ matmul_i1_avx2 (gfc_array_m1 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_UINTEGER_1)
+      && (axstride_bytes == sizeof (GFC_UINTEGER_1)
+	  || aystride_bytes == sizeof (GFC_UINTEGER_1))
+      && (bxstride_bytes == sizeof (GFC_UINTEGER_1)
+	  || bystride_bytes == sizeof (GFC_UINTEGER_1))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_UINTEGER_1 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_UINTEGER_1)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_UINTEGER_1)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -859,12 +869,12 @@ matmul_i1_avx2 (gfc_array_m1 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_UINTEGER_1) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_UINTEGER_1) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -873,7 +883,9 @@ matmul_i1_avx2 (gfc_array_m1 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_UINTEGER_1)
+      && axstride_bytes == sizeof (GFC_UINTEGER_1)
+      && bxstride_bytes == sizeof (GFC_UINTEGER_1)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -926,7 +938,7 @@ matmul_i1_avx2 (gfc_array_m1 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_UINTEGER_1))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -1152,7 +1164,9 @@ matmul_i1_avx2 (gfc_array_m1 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_UINTEGER_1)
+	   && aystride_bytes == sizeof (GFC_UINTEGER_1)
+	   && bxstride_bytes == sizeof (GFC_UINTEGER_1))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -1205,7 +1219,7 @@ matmul_i1_avx2 (gfc_array_m1 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -1262,7 +1276,7 @@ matmul_i1_avx512f (gfc_array_m1 * const restrict retarray,
   const GFC_UINTEGER_1 * restrict bbase;
   GFC_UINTEGER_1 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -1353,12 +1367,11 @@ matmul_i1_avx512f (gfc_array_m1 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -1427,15 +1440,19 @@ matmul_i1_avx512f (gfc_array_m1 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_UINTEGER_1)
+      && (axstride_bytes == sizeof (GFC_UINTEGER_1)
+	  || aystride_bytes == sizeof (GFC_UINTEGER_1))
+      && (bxstride_bytes == sizeof (GFC_UINTEGER_1)
+	  || bystride_bytes == sizeof (GFC_UINTEGER_1))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_UINTEGER_1 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_UINTEGER_1)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_UINTEGER_1)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -1444,12 +1461,12 @@ matmul_i1_avx512f (gfc_array_m1 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_UINTEGER_1) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_UINTEGER_1) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -1458,7 +1475,9 @@ matmul_i1_avx512f (gfc_array_m1 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_UINTEGER_1)
+      && axstride_bytes == sizeof (GFC_UINTEGER_1)
+      && bxstride_bytes == sizeof (GFC_UINTEGER_1)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -1511,7 +1530,7 @@ matmul_i1_avx512f (gfc_array_m1 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_UINTEGER_1))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -1737,7 +1756,9 @@ matmul_i1_avx512f (gfc_array_m1 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_UINTEGER_1)
+	   && aystride_bytes == sizeof (GFC_UINTEGER_1)
+	   && bxstride_bytes == sizeof (GFC_UINTEGER_1))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -1790,7 +1811,7 @@ matmul_i1_avx512f (gfc_array_m1 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -1861,7 +1882,7 @@ matmul_i1_vanilla (gfc_array_m1 * const restrict retarray,
   const GFC_UINTEGER_1 * restrict bbase;
   GFC_UINTEGER_1 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -1952,12 +1973,11 @@ matmul_i1_vanilla (gfc_array_m1 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -2026,15 +2046,19 @@ matmul_i1_vanilla (gfc_array_m1 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_UINTEGER_1)
+      && (axstride_bytes == sizeof (GFC_UINTEGER_1)
+	  || aystride_bytes == sizeof (GFC_UINTEGER_1))
+      && (bxstride_bytes == sizeof (GFC_UINTEGER_1)
+	  || bystride_bytes == sizeof (GFC_UINTEGER_1))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_UINTEGER_1 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_UINTEGER_1)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_UINTEGER_1)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -2043,12 +2067,12 @@ matmul_i1_vanilla (gfc_array_m1 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_UINTEGER_1) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_UINTEGER_1) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -2057,7 +2081,9 @@ matmul_i1_vanilla (gfc_array_m1 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_UINTEGER_1)
+      && axstride_bytes == sizeof (GFC_UINTEGER_1)
+      && bxstride_bytes == sizeof (GFC_UINTEGER_1)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -2110,7 +2136,7 @@ matmul_i1_vanilla (gfc_array_m1 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_UINTEGER_1))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -2336,7 +2362,9 @@ matmul_i1_vanilla (gfc_array_m1 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_UINTEGER_1)
+	   && aystride_bytes == sizeof (GFC_UINTEGER_1)
+	   && bxstride_bytes == sizeof (GFC_UINTEGER_1))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -2389,7 +2417,7 @@ matmul_i1_vanilla (gfc_array_m1 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -2519,7 +2547,7 @@ matmul_i1 (gfc_array_m1 * const restrict retarray,
   const GFC_UINTEGER_1 * restrict bbase;
   GFC_UINTEGER_1 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -2610,12 +2638,11 @@ matmul_i1 (gfc_array_m1 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -2684,15 +2711,19 @@ matmul_i1 (gfc_array_m1 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_UINTEGER_1)
+      && (axstride_bytes == sizeof (GFC_UINTEGER_1)
+	  || aystride_bytes == sizeof (GFC_UINTEGER_1))
+      && (bxstride_bytes == sizeof (GFC_UINTEGER_1)
+	  || bystride_bytes == sizeof (GFC_UINTEGER_1))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_UINTEGER_1 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_UINTEGER_1)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_UINTEGER_1)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -2701,12 +2732,12 @@ matmul_i1 (gfc_array_m1 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_UINTEGER_1) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_UINTEGER_1) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -2715,7 +2746,9 @@ matmul_i1 (gfc_array_m1 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_UINTEGER_1)
+      && axstride_bytes == sizeof (GFC_UINTEGER_1)
+      && bxstride_bytes == sizeof (GFC_UINTEGER_1)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -2768,7 +2801,7 @@ matmul_i1 (gfc_array_m1 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_UINTEGER_1))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -2994,7 +3027,9 @@ matmul_i1 (gfc_array_m1 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_UINTEGER_1)
+	   && aystride_bytes == sizeof (GFC_UINTEGER_1)
+	   && bxstride_bytes == sizeof (GFC_UINTEGER_1))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -3047,7 +3082,7 @@ matmul_i1 (gfc_array_m1 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
diff --git a/libgfortran/generated/matmul_i16.c b/libgfortran/generated/matmul_i16.c
index aedf424f933f..7518a890ea17 100644
--- a/libgfortran/generated/matmul_i16.c
+++ b/libgfortran/generated/matmul_i16.c
@@ -92,7 +92,7 @@ matmul_i16_avx (gfc_array_m16 * const restrict retarray,
   const GFC_UINTEGER_16 * restrict bbase;
   GFC_UINTEGER_16 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -183,12 +183,11 @@ matmul_i16_avx (gfc_array_m16 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -257,15 +256,19 @@ matmul_i16_avx (gfc_array_m16 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_UINTEGER_16)
+      && (axstride_bytes == sizeof (GFC_UINTEGER_16)
+	  || aystride_bytes == sizeof (GFC_UINTEGER_16))
+      && (bxstride_bytes == sizeof (GFC_UINTEGER_16)
+	  || bystride_bytes == sizeof (GFC_UINTEGER_16))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_UINTEGER_16 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_UINTEGER_16)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_UINTEGER_16)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -274,12 +277,12 @@ matmul_i16_avx (gfc_array_m16 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_UINTEGER_16) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_UINTEGER_16) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -288,7 +291,9 @@ matmul_i16_avx (gfc_array_m16 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_UINTEGER_16)
+      && axstride_bytes == sizeof (GFC_UINTEGER_16)
+      && bxstride_bytes == sizeof (GFC_UINTEGER_16)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -341,7 +346,7 @@ matmul_i16_avx (gfc_array_m16 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_UINTEGER_16))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -567,7 +572,9 @@ matmul_i16_avx (gfc_array_m16 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_UINTEGER_16)
+	   && aystride_bytes == sizeof (GFC_UINTEGER_16)
+	   && bxstride_bytes == sizeof (GFC_UINTEGER_16))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -620,7 +627,7 @@ matmul_i16_avx (gfc_array_m16 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -677,7 +684,7 @@ matmul_i16_avx2 (gfc_array_m16 * const restrict retarray,
   const GFC_UINTEGER_16 * restrict bbase;
   GFC_UINTEGER_16 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -768,12 +775,11 @@ matmul_i16_avx2 (gfc_array_m16 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -842,15 +848,19 @@ matmul_i16_avx2 (gfc_array_m16 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_UINTEGER_16)
+      && (axstride_bytes == sizeof (GFC_UINTEGER_16)
+	  || aystride_bytes == sizeof (GFC_UINTEGER_16))
+      && (bxstride_bytes == sizeof (GFC_UINTEGER_16)
+	  || bystride_bytes == sizeof (GFC_UINTEGER_16))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_UINTEGER_16 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_UINTEGER_16)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_UINTEGER_16)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -859,12 +869,12 @@ matmul_i16_avx2 (gfc_array_m16 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_UINTEGER_16) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_UINTEGER_16) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -873,7 +883,9 @@ matmul_i16_avx2 (gfc_array_m16 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_UINTEGER_16)
+      && axstride_bytes == sizeof (GFC_UINTEGER_16)
+      && bxstride_bytes == sizeof (GFC_UINTEGER_16)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -926,7 +938,7 @@ matmul_i16_avx2 (gfc_array_m16 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_UINTEGER_16))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -1152,7 +1164,9 @@ matmul_i16_avx2 (gfc_array_m16 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_UINTEGER_16)
+	   && aystride_bytes == sizeof (GFC_UINTEGER_16)
+	   && bxstride_bytes == sizeof (GFC_UINTEGER_16))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -1205,7 +1219,7 @@ matmul_i16_avx2 (gfc_array_m16 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -1262,7 +1276,7 @@ matmul_i16_avx512f (gfc_array_m16 * const restrict retarray,
   const GFC_UINTEGER_16 * restrict bbase;
   GFC_UINTEGER_16 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -1353,12 +1367,11 @@ matmul_i16_avx512f (gfc_array_m16 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -1427,15 +1440,19 @@ matmul_i16_avx512f (gfc_array_m16 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_UINTEGER_16)
+      && (axstride_bytes == sizeof (GFC_UINTEGER_16)
+	  || aystride_bytes == sizeof (GFC_UINTEGER_16))
+      && (bxstride_bytes == sizeof (GFC_UINTEGER_16)
+	  || bystride_bytes == sizeof (GFC_UINTEGER_16))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_UINTEGER_16 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_UINTEGER_16)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_UINTEGER_16)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -1444,12 +1461,12 @@ matmul_i16_avx512f (gfc_array_m16 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_UINTEGER_16) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_UINTEGER_16) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -1458,7 +1475,9 @@ matmul_i16_avx512f (gfc_array_m16 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_UINTEGER_16)
+      && axstride_bytes == sizeof (GFC_UINTEGER_16)
+      && bxstride_bytes == sizeof (GFC_UINTEGER_16)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -1511,7 +1530,7 @@ matmul_i16_avx512f (gfc_array_m16 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_UINTEGER_16))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -1737,7 +1756,9 @@ matmul_i16_avx512f (gfc_array_m16 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_UINTEGER_16)
+	   && aystride_bytes == sizeof (GFC_UINTEGER_16)
+	   && bxstride_bytes == sizeof (GFC_UINTEGER_16))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -1790,7 +1811,7 @@ matmul_i16_avx512f (gfc_array_m16 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -1861,7 +1882,7 @@ matmul_i16_vanilla (gfc_array_m16 * const restrict retarray,
   const GFC_UINTEGER_16 * restrict bbase;
   GFC_UINTEGER_16 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -1952,12 +1973,11 @@ matmul_i16_vanilla (gfc_array_m16 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -2026,15 +2046,19 @@ matmul_i16_vanilla (gfc_array_m16 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_UINTEGER_16)
+      && (axstride_bytes == sizeof (GFC_UINTEGER_16)
+	  || aystride_bytes == sizeof (GFC_UINTEGER_16))
+      && (bxstride_bytes == sizeof (GFC_UINTEGER_16)
+	  || bystride_bytes == sizeof (GFC_UINTEGER_16))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_UINTEGER_16 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_UINTEGER_16)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_UINTEGER_16)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -2043,12 +2067,12 @@ matmul_i16_vanilla (gfc_array_m16 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_UINTEGER_16) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_UINTEGER_16) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -2057,7 +2081,9 @@ matmul_i16_vanilla (gfc_array_m16 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_UINTEGER_16)
+      && axstride_bytes == sizeof (GFC_UINTEGER_16)
+      && bxstride_bytes == sizeof (GFC_UINTEGER_16)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -2110,7 +2136,7 @@ matmul_i16_vanilla (gfc_array_m16 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_UINTEGER_16))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -2336,7 +2362,9 @@ matmul_i16_vanilla (gfc_array_m16 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_UINTEGER_16)
+	   && aystride_bytes == sizeof (GFC_UINTEGER_16)
+	   && bxstride_bytes == sizeof (GFC_UINTEGER_16))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -2389,7 +2417,7 @@ matmul_i16_vanilla (gfc_array_m16 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -2519,7 +2547,7 @@ matmul_i16 (gfc_array_m16 * const restrict retarray,
   const GFC_UINTEGER_16 * restrict bbase;
   GFC_UINTEGER_16 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -2610,12 +2638,11 @@ matmul_i16 (gfc_array_m16 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -2684,15 +2711,19 @@ matmul_i16 (gfc_array_m16 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_UINTEGER_16)
+      && (axstride_bytes == sizeof (GFC_UINTEGER_16)
+	  || aystride_bytes == sizeof (GFC_UINTEGER_16))
+      && (bxstride_bytes == sizeof (GFC_UINTEGER_16)
+	  || bystride_bytes == sizeof (GFC_UINTEGER_16))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_UINTEGER_16 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_UINTEGER_16)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_UINTEGER_16)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -2701,12 +2732,12 @@ matmul_i16 (gfc_array_m16 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_UINTEGER_16) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_UINTEGER_16) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -2715,7 +2746,9 @@ matmul_i16 (gfc_array_m16 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_UINTEGER_16)
+      && axstride_bytes == sizeof (GFC_UINTEGER_16)
+      && bxstride_bytes == sizeof (GFC_UINTEGER_16)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -2768,7 +2801,7 @@ matmul_i16 (gfc_array_m16 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_UINTEGER_16))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -2994,7 +3027,9 @@ matmul_i16 (gfc_array_m16 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_UINTEGER_16)
+	   && aystride_bytes == sizeof (GFC_UINTEGER_16)
+	   && bxstride_bytes == sizeof (GFC_UINTEGER_16))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -3047,7 +3082,7 @@ matmul_i16 (gfc_array_m16 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
diff --git a/libgfortran/generated/matmul_i2.c b/libgfortran/generated/matmul_i2.c
index 2dc463c6ff67..81cbd11362d3 100644
--- a/libgfortran/generated/matmul_i2.c
+++ b/libgfortran/generated/matmul_i2.c
@@ -92,7 +92,7 @@ matmul_i2_avx (gfc_array_m2 * const restrict retarray,
   const GFC_UINTEGER_2 * restrict bbase;
   GFC_UINTEGER_2 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -183,12 +183,11 @@ matmul_i2_avx (gfc_array_m2 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -257,15 +256,19 @@ matmul_i2_avx (gfc_array_m2 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_UINTEGER_2)
+      && (axstride_bytes == sizeof (GFC_UINTEGER_2)
+	  || aystride_bytes == sizeof (GFC_UINTEGER_2))
+      && (bxstride_bytes == sizeof (GFC_UINTEGER_2)
+	  || bystride_bytes == sizeof (GFC_UINTEGER_2))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_UINTEGER_2 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_UINTEGER_2)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_UINTEGER_2)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -274,12 +277,12 @@ matmul_i2_avx (gfc_array_m2 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_UINTEGER_2) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_UINTEGER_2) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -288,7 +291,9 @@ matmul_i2_avx (gfc_array_m2 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_UINTEGER_2)
+      && axstride_bytes == sizeof (GFC_UINTEGER_2)
+      && bxstride_bytes == sizeof (GFC_UINTEGER_2)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -341,7 +346,7 @@ matmul_i2_avx (gfc_array_m2 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_UINTEGER_2))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -567,7 +572,9 @@ matmul_i2_avx (gfc_array_m2 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_UINTEGER_2)
+	   && aystride_bytes == sizeof (GFC_UINTEGER_2)
+	   && bxstride_bytes == sizeof (GFC_UINTEGER_2))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -620,7 +627,7 @@ matmul_i2_avx (gfc_array_m2 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -677,7 +684,7 @@ matmul_i2_avx2 (gfc_array_m2 * const restrict retarray,
   const GFC_UINTEGER_2 * restrict bbase;
   GFC_UINTEGER_2 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -768,12 +775,11 @@ matmul_i2_avx2 (gfc_array_m2 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -842,15 +848,19 @@ matmul_i2_avx2 (gfc_array_m2 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_UINTEGER_2)
+      && (axstride_bytes == sizeof (GFC_UINTEGER_2)
+	  || aystride_bytes == sizeof (GFC_UINTEGER_2))
+      && (bxstride_bytes == sizeof (GFC_UINTEGER_2)
+	  || bystride_bytes == sizeof (GFC_UINTEGER_2))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_UINTEGER_2 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_UINTEGER_2)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_UINTEGER_2)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -859,12 +869,12 @@ matmul_i2_avx2 (gfc_array_m2 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_UINTEGER_2) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_UINTEGER_2) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -873,7 +883,9 @@ matmul_i2_avx2 (gfc_array_m2 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_UINTEGER_2)
+      && axstride_bytes == sizeof (GFC_UINTEGER_2)
+      && bxstride_bytes == sizeof (GFC_UINTEGER_2)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -926,7 +938,7 @@ matmul_i2_avx2 (gfc_array_m2 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_UINTEGER_2))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -1152,7 +1164,9 @@ matmul_i2_avx2 (gfc_array_m2 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_UINTEGER_2)
+	   && aystride_bytes == sizeof (GFC_UINTEGER_2)
+	   && bxstride_bytes == sizeof (GFC_UINTEGER_2))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -1205,7 +1219,7 @@ matmul_i2_avx2 (gfc_array_m2 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -1262,7 +1276,7 @@ matmul_i2_avx512f (gfc_array_m2 * const restrict retarray,
   const GFC_UINTEGER_2 * restrict bbase;
   GFC_UINTEGER_2 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -1353,12 +1367,11 @@ matmul_i2_avx512f (gfc_array_m2 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -1427,15 +1440,19 @@ matmul_i2_avx512f (gfc_array_m2 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_UINTEGER_2)
+      && (axstride_bytes == sizeof (GFC_UINTEGER_2)
+	  || aystride_bytes == sizeof (GFC_UINTEGER_2))
+      && (bxstride_bytes == sizeof (GFC_UINTEGER_2)
+	  || bystride_bytes == sizeof (GFC_UINTEGER_2))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_UINTEGER_2 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_UINTEGER_2)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_UINTEGER_2)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -1444,12 +1461,12 @@ matmul_i2_avx512f (gfc_array_m2 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_UINTEGER_2) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_UINTEGER_2) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -1458,7 +1475,9 @@ matmul_i2_avx512f (gfc_array_m2 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_UINTEGER_2)
+      && axstride_bytes == sizeof (GFC_UINTEGER_2)
+      && bxstride_bytes == sizeof (GFC_UINTEGER_2)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -1511,7 +1530,7 @@ matmul_i2_avx512f (gfc_array_m2 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_UINTEGER_2))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -1737,7 +1756,9 @@ matmul_i2_avx512f (gfc_array_m2 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_UINTEGER_2)
+	   && aystride_bytes == sizeof (GFC_UINTEGER_2)
+	   && bxstride_bytes == sizeof (GFC_UINTEGER_2))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -1790,7 +1811,7 @@ matmul_i2_avx512f (gfc_array_m2 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -1861,7 +1882,7 @@ matmul_i2_vanilla (gfc_array_m2 * const restrict retarray,
   const GFC_UINTEGER_2 * restrict bbase;
   GFC_UINTEGER_2 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -1952,12 +1973,11 @@ matmul_i2_vanilla (gfc_array_m2 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -2026,15 +2046,19 @@ matmul_i2_vanilla (gfc_array_m2 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_UINTEGER_2)
+      && (axstride_bytes == sizeof (GFC_UINTEGER_2)
+	  || aystride_bytes == sizeof (GFC_UINTEGER_2))
+      && (bxstride_bytes == sizeof (GFC_UINTEGER_2)
+	  || bystride_bytes == sizeof (GFC_UINTEGER_2))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_UINTEGER_2 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_UINTEGER_2)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_UINTEGER_2)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -2043,12 +2067,12 @@ matmul_i2_vanilla (gfc_array_m2 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_UINTEGER_2) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_UINTEGER_2) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -2057,7 +2081,9 @@ matmul_i2_vanilla (gfc_array_m2 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_UINTEGER_2)
+      && axstride_bytes == sizeof (GFC_UINTEGER_2)
+      && bxstride_bytes == sizeof (GFC_UINTEGER_2)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -2110,7 +2136,7 @@ matmul_i2_vanilla (gfc_array_m2 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_UINTEGER_2))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -2336,7 +2362,9 @@ matmul_i2_vanilla (gfc_array_m2 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_UINTEGER_2)
+	   && aystride_bytes == sizeof (GFC_UINTEGER_2)
+	   && bxstride_bytes == sizeof (GFC_UINTEGER_2))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -2389,7 +2417,7 @@ matmul_i2_vanilla (gfc_array_m2 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -2519,7 +2547,7 @@ matmul_i2 (gfc_array_m2 * const restrict retarray,
   const GFC_UINTEGER_2 * restrict bbase;
   GFC_UINTEGER_2 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -2610,12 +2638,11 @@ matmul_i2 (gfc_array_m2 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -2684,15 +2711,19 @@ matmul_i2 (gfc_array_m2 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_UINTEGER_2)
+      && (axstride_bytes == sizeof (GFC_UINTEGER_2)
+	  || aystride_bytes == sizeof (GFC_UINTEGER_2))
+      && (bxstride_bytes == sizeof (GFC_UINTEGER_2)
+	  || bystride_bytes == sizeof (GFC_UINTEGER_2))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_UINTEGER_2 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_UINTEGER_2)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_UINTEGER_2)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -2701,12 +2732,12 @@ matmul_i2 (gfc_array_m2 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_UINTEGER_2) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_UINTEGER_2) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -2715,7 +2746,9 @@ matmul_i2 (gfc_array_m2 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_UINTEGER_2)
+      && axstride_bytes == sizeof (GFC_UINTEGER_2)
+      && bxstride_bytes == sizeof (GFC_UINTEGER_2)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -2768,7 +2801,7 @@ matmul_i2 (gfc_array_m2 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_UINTEGER_2))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -2994,7 +3027,9 @@ matmul_i2 (gfc_array_m2 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_UINTEGER_2)
+	   && aystride_bytes == sizeof (GFC_UINTEGER_2)
+	   && bxstride_bytes == sizeof (GFC_UINTEGER_2))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -3047,7 +3082,7 @@ matmul_i2 (gfc_array_m2 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
diff --git a/libgfortran/generated/matmul_i4.c b/libgfortran/generated/matmul_i4.c
index ca9b2dc41d7d..cc6ab5fe326c 100644
--- a/libgfortran/generated/matmul_i4.c
+++ b/libgfortran/generated/matmul_i4.c
@@ -92,7 +92,7 @@ matmul_i4_avx (gfc_array_m4 * const restrict retarray,
   const GFC_UINTEGER_4 * restrict bbase;
   GFC_UINTEGER_4 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -183,12 +183,11 @@ matmul_i4_avx (gfc_array_m4 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -257,15 +256,19 @@ matmul_i4_avx (gfc_array_m4 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_UINTEGER_4)
+      && (axstride_bytes == sizeof (GFC_UINTEGER_4)
+	  || aystride_bytes == sizeof (GFC_UINTEGER_4))
+      && (bxstride_bytes == sizeof (GFC_UINTEGER_4)
+	  || bystride_bytes == sizeof (GFC_UINTEGER_4))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_UINTEGER_4 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_UINTEGER_4)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_UINTEGER_4)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -274,12 +277,12 @@ matmul_i4_avx (gfc_array_m4 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_UINTEGER_4) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_UINTEGER_4) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -288,7 +291,9 @@ matmul_i4_avx (gfc_array_m4 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_UINTEGER_4)
+      && axstride_bytes == sizeof (GFC_UINTEGER_4)
+      && bxstride_bytes == sizeof (GFC_UINTEGER_4)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -341,7 +346,7 @@ matmul_i4_avx (gfc_array_m4 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_UINTEGER_4))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -567,7 +572,9 @@ matmul_i4_avx (gfc_array_m4 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_UINTEGER_4)
+	   && aystride_bytes == sizeof (GFC_UINTEGER_4)
+	   && bxstride_bytes == sizeof (GFC_UINTEGER_4))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -620,7 +627,7 @@ matmul_i4_avx (gfc_array_m4 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -677,7 +684,7 @@ matmul_i4_avx2 (gfc_array_m4 * const restrict retarray,
   const GFC_UINTEGER_4 * restrict bbase;
   GFC_UINTEGER_4 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -768,12 +775,11 @@ matmul_i4_avx2 (gfc_array_m4 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -842,15 +848,19 @@ matmul_i4_avx2 (gfc_array_m4 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_UINTEGER_4)
+      && (axstride_bytes == sizeof (GFC_UINTEGER_4)
+	  || aystride_bytes == sizeof (GFC_UINTEGER_4))
+      && (bxstride_bytes == sizeof (GFC_UINTEGER_4)
+	  || bystride_bytes == sizeof (GFC_UINTEGER_4))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_UINTEGER_4 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_UINTEGER_4)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_UINTEGER_4)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -859,12 +869,12 @@ matmul_i4_avx2 (gfc_array_m4 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_UINTEGER_4) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_UINTEGER_4) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -873,7 +883,9 @@ matmul_i4_avx2 (gfc_array_m4 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_UINTEGER_4)
+      && axstride_bytes == sizeof (GFC_UINTEGER_4)
+      && bxstride_bytes == sizeof (GFC_UINTEGER_4)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -926,7 +938,7 @@ matmul_i4_avx2 (gfc_array_m4 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_UINTEGER_4))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -1152,7 +1164,9 @@ matmul_i4_avx2 (gfc_array_m4 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_UINTEGER_4)
+	   && aystride_bytes == sizeof (GFC_UINTEGER_4)
+	   && bxstride_bytes == sizeof (GFC_UINTEGER_4))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -1205,7 +1219,7 @@ matmul_i4_avx2 (gfc_array_m4 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -1262,7 +1276,7 @@ matmul_i4_avx512f (gfc_array_m4 * const restrict retarray,
   const GFC_UINTEGER_4 * restrict bbase;
   GFC_UINTEGER_4 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -1353,12 +1367,11 @@ matmul_i4_avx512f (gfc_array_m4 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -1427,15 +1440,19 @@ matmul_i4_avx512f (gfc_array_m4 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_UINTEGER_4)
+      && (axstride_bytes == sizeof (GFC_UINTEGER_4)
+	  || aystride_bytes == sizeof (GFC_UINTEGER_4))
+      && (bxstride_bytes == sizeof (GFC_UINTEGER_4)
+	  || bystride_bytes == sizeof (GFC_UINTEGER_4))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_UINTEGER_4 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_UINTEGER_4)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_UINTEGER_4)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -1444,12 +1461,12 @@ matmul_i4_avx512f (gfc_array_m4 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_UINTEGER_4) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_UINTEGER_4) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -1458,7 +1475,9 @@ matmul_i4_avx512f (gfc_array_m4 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_UINTEGER_4)
+      && axstride_bytes == sizeof (GFC_UINTEGER_4)
+      && bxstride_bytes == sizeof (GFC_UINTEGER_4)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -1511,7 +1530,7 @@ matmul_i4_avx512f (gfc_array_m4 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_UINTEGER_4))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -1737,7 +1756,9 @@ matmul_i4_avx512f (gfc_array_m4 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_UINTEGER_4)
+	   && aystride_bytes == sizeof (GFC_UINTEGER_4)
+	   && bxstride_bytes == sizeof (GFC_UINTEGER_4))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -1790,7 +1811,7 @@ matmul_i4_avx512f (gfc_array_m4 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -1861,7 +1882,7 @@ matmul_i4_vanilla (gfc_array_m4 * const restrict retarray,
   const GFC_UINTEGER_4 * restrict bbase;
   GFC_UINTEGER_4 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -1952,12 +1973,11 @@ matmul_i4_vanilla (gfc_array_m4 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -2026,15 +2046,19 @@ matmul_i4_vanilla (gfc_array_m4 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_UINTEGER_4)
+      && (axstride_bytes == sizeof (GFC_UINTEGER_4)
+	  || aystride_bytes == sizeof (GFC_UINTEGER_4))
+      && (bxstride_bytes == sizeof (GFC_UINTEGER_4)
+	  || bystride_bytes == sizeof (GFC_UINTEGER_4))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_UINTEGER_4 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_UINTEGER_4)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_UINTEGER_4)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -2043,12 +2067,12 @@ matmul_i4_vanilla (gfc_array_m4 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_UINTEGER_4) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_UINTEGER_4) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -2057,7 +2081,9 @@ matmul_i4_vanilla (gfc_array_m4 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_UINTEGER_4)
+      && axstride_bytes == sizeof (GFC_UINTEGER_4)
+      && bxstride_bytes == sizeof (GFC_UINTEGER_4)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -2110,7 +2136,7 @@ matmul_i4_vanilla (gfc_array_m4 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_UINTEGER_4))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -2336,7 +2362,9 @@ matmul_i4_vanilla (gfc_array_m4 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_UINTEGER_4)
+	   && aystride_bytes == sizeof (GFC_UINTEGER_4)
+	   && bxstride_bytes == sizeof (GFC_UINTEGER_4))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -2389,7 +2417,7 @@ matmul_i4_vanilla (gfc_array_m4 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -2519,7 +2547,7 @@ matmul_i4 (gfc_array_m4 * const restrict retarray,
   const GFC_UINTEGER_4 * restrict bbase;
   GFC_UINTEGER_4 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -2610,12 +2638,11 @@ matmul_i4 (gfc_array_m4 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -2684,15 +2711,19 @@ matmul_i4 (gfc_array_m4 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_UINTEGER_4)
+      && (axstride_bytes == sizeof (GFC_UINTEGER_4)
+	  || aystride_bytes == sizeof (GFC_UINTEGER_4))
+      && (bxstride_bytes == sizeof (GFC_UINTEGER_4)
+	  || bystride_bytes == sizeof (GFC_UINTEGER_4))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_UINTEGER_4 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_UINTEGER_4)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_UINTEGER_4)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -2701,12 +2732,12 @@ matmul_i4 (gfc_array_m4 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_UINTEGER_4) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_UINTEGER_4) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -2715,7 +2746,9 @@ matmul_i4 (gfc_array_m4 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_UINTEGER_4)
+      && axstride_bytes == sizeof (GFC_UINTEGER_4)
+      && bxstride_bytes == sizeof (GFC_UINTEGER_4)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -2768,7 +2801,7 @@ matmul_i4 (gfc_array_m4 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_UINTEGER_4))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -2994,7 +3027,9 @@ matmul_i4 (gfc_array_m4 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_UINTEGER_4)
+	   && aystride_bytes == sizeof (GFC_UINTEGER_4)
+	   && bxstride_bytes == sizeof (GFC_UINTEGER_4))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -3047,7 +3082,7 @@ matmul_i4 (gfc_array_m4 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
diff --git a/libgfortran/generated/matmul_i8.c b/libgfortran/generated/matmul_i8.c
index a8da3826b658..a5e8f39b8327 100644
--- a/libgfortran/generated/matmul_i8.c
+++ b/libgfortran/generated/matmul_i8.c
@@ -92,7 +92,7 @@ matmul_i8_avx (gfc_array_m8 * const restrict retarray,
   const GFC_UINTEGER_8 * restrict bbase;
   GFC_UINTEGER_8 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -183,12 +183,11 @@ matmul_i8_avx (gfc_array_m8 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -257,15 +256,19 @@ matmul_i8_avx (gfc_array_m8 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_UINTEGER_8)
+      && (axstride_bytes == sizeof (GFC_UINTEGER_8)
+	  || aystride_bytes == sizeof (GFC_UINTEGER_8))
+      && (bxstride_bytes == sizeof (GFC_UINTEGER_8)
+	  || bystride_bytes == sizeof (GFC_UINTEGER_8))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_UINTEGER_8 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_UINTEGER_8)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_UINTEGER_8)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -274,12 +277,12 @@ matmul_i8_avx (gfc_array_m8 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_UINTEGER_8) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_UINTEGER_8) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -288,7 +291,9 @@ matmul_i8_avx (gfc_array_m8 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_UINTEGER_8)
+      && axstride_bytes == sizeof (GFC_UINTEGER_8)
+      && bxstride_bytes == sizeof (GFC_UINTEGER_8)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -341,7 +346,7 @@ matmul_i8_avx (gfc_array_m8 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_UINTEGER_8))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -567,7 +572,9 @@ matmul_i8_avx (gfc_array_m8 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_UINTEGER_8)
+	   && aystride_bytes == sizeof (GFC_UINTEGER_8)
+	   && bxstride_bytes == sizeof (GFC_UINTEGER_8))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -620,7 +627,7 @@ matmul_i8_avx (gfc_array_m8 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -677,7 +684,7 @@ matmul_i8_avx2 (gfc_array_m8 * const restrict retarray,
   const GFC_UINTEGER_8 * restrict bbase;
   GFC_UINTEGER_8 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -768,12 +775,11 @@ matmul_i8_avx2 (gfc_array_m8 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -842,15 +848,19 @@ matmul_i8_avx2 (gfc_array_m8 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_UINTEGER_8)
+      && (axstride_bytes == sizeof (GFC_UINTEGER_8)
+	  || aystride_bytes == sizeof (GFC_UINTEGER_8))
+      && (bxstride_bytes == sizeof (GFC_UINTEGER_8)
+	  || bystride_bytes == sizeof (GFC_UINTEGER_8))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_UINTEGER_8 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_UINTEGER_8)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_UINTEGER_8)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -859,12 +869,12 @@ matmul_i8_avx2 (gfc_array_m8 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_UINTEGER_8) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_UINTEGER_8) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -873,7 +883,9 @@ matmul_i8_avx2 (gfc_array_m8 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_UINTEGER_8)
+      && axstride_bytes == sizeof (GFC_UINTEGER_8)
+      && bxstride_bytes == sizeof (GFC_UINTEGER_8)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -926,7 +938,7 @@ matmul_i8_avx2 (gfc_array_m8 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_UINTEGER_8))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -1152,7 +1164,9 @@ matmul_i8_avx2 (gfc_array_m8 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_UINTEGER_8)
+	   && aystride_bytes == sizeof (GFC_UINTEGER_8)
+	   && bxstride_bytes == sizeof (GFC_UINTEGER_8))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -1205,7 +1219,7 @@ matmul_i8_avx2 (gfc_array_m8 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -1262,7 +1276,7 @@ matmul_i8_avx512f (gfc_array_m8 * const restrict retarray,
   const GFC_UINTEGER_8 * restrict bbase;
   GFC_UINTEGER_8 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -1353,12 +1367,11 @@ matmul_i8_avx512f (gfc_array_m8 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -1427,15 +1440,19 @@ matmul_i8_avx512f (gfc_array_m8 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_UINTEGER_8)
+      && (axstride_bytes == sizeof (GFC_UINTEGER_8)
+	  || aystride_bytes == sizeof (GFC_UINTEGER_8))
+      && (bxstride_bytes == sizeof (GFC_UINTEGER_8)
+	  || bystride_bytes == sizeof (GFC_UINTEGER_8))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_UINTEGER_8 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_UINTEGER_8)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_UINTEGER_8)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -1444,12 +1461,12 @@ matmul_i8_avx512f (gfc_array_m8 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_UINTEGER_8) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_UINTEGER_8) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -1458,7 +1475,9 @@ matmul_i8_avx512f (gfc_array_m8 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_UINTEGER_8)
+      && axstride_bytes == sizeof (GFC_UINTEGER_8)
+      && bxstride_bytes == sizeof (GFC_UINTEGER_8)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -1511,7 +1530,7 @@ matmul_i8_avx512f (gfc_array_m8 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_UINTEGER_8))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -1737,7 +1756,9 @@ matmul_i8_avx512f (gfc_array_m8 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_UINTEGER_8)
+	   && aystride_bytes == sizeof (GFC_UINTEGER_8)
+	   && bxstride_bytes == sizeof (GFC_UINTEGER_8))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -1790,7 +1811,7 @@ matmul_i8_avx512f (gfc_array_m8 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -1861,7 +1882,7 @@ matmul_i8_vanilla (gfc_array_m8 * const restrict retarray,
   const GFC_UINTEGER_8 * restrict bbase;
   GFC_UINTEGER_8 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -1952,12 +1973,11 @@ matmul_i8_vanilla (gfc_array_m8 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -2026,15 +2046,19 @@ matmul_i8_vanilla (gfc_array_m8 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_UINTEGER_8)
+      && (axstride_bytes == sizeof (GFC_UINTEGER_8)
+	  || aystride_bytes == sizeof (GFC_UINTEGER_8))
+      && (bxstride_bytes == sizeof (GFC_UINTEGER_8)
+	  || bystride_bytes == sizeof (GFC_UINTEGER_8))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_UINTEGER_8 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_UINTEGER_8)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_UINTEGER_8)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -2043,12 +2067,12 @@ matmul_i8_vanilla (gfc_array_m8 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_UINTEGER_8) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_UINTEGER_8) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -2057,7 +2081,9 @@ matmul_i8_vanilla (gfc_array_m8 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_UINTEGER_8)
+      && axstride_bytes == sizeof (GFC_UINTEGER_8)
+      && bxstride_bytes == sizeof (GFC_UINTEGER_8)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -2110,7 +2136,7 @@ matmul_i8_vanilla (gfc_array_m8 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_UINTEGER_8))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -2336,7 +2362,9 @@ matmul_i8_vanilla (gfc_array_m8 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_UINTEGER_8)
+	   && aystride_bytes == sizeof (GFC_UINTEGER_8)
+	   && bxstride_bytes == sizeof (GFC_UINTEGER_8))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -2389,7 +2417,7 @@ matmul_i8_vanilla (gfc_array_m8 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -2519,7 +2547,7 @@ matmul_i8 (gfc_array_m8 * const restrict retarray,
   const GFC_UINTEGER_8 * restrict bbase;
   GFC_UINTEGER_8 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -2610,12 +2638,11 @@ matmul_i8 (gfc_array_m8 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -2684,15 +2711,19 @@ matmul_i8 (gfc_array_m8 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_UINTEGER_8)
+      && (axstride_bytes == sizeof (GFC_UINTEGER_8)
+	  || aystride_bytes == sizeof (GFC_UINTEGER_8))
+      && (bxstride_bytes == sizeof (GFC_UINTEGER_8)
+	  || bystride_bytes == sizeof (GFC_UINTEGER_8))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_UINTEGER_8 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_UINTEGER_8)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_UINTEGER_8)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -2701,12 +2732,12 @@ matmul_i8 (gfc_array_m8 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_UINTEGER_8) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_UINTEGER_8) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -2715,7 +2746,9 @@ matmul_i8 (gfc_array_m8 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_UINTEGER_8)
+      && axstride_bytes == sizeof (GFC_UINTEGER_8)
+      && bxstride_bytes == sizeof (GFC_UINTEGER_8)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -2768,7 +2801,7 @@ matmul_i8 (gfc_array_m8 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_UINTEGER_8))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -2994,7 +3027,9 @@ matmul_i8 (gfc_array_m8 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_UINTEGER_8)
+	   && aystride_bytes == sizeof (GFC_UINTEGER_8)
+	   && bxstride_bytes == sizeof (GFC_UINTEGER_8))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -3047,7 +3082,7 @@ matmul_i8 (gfc_array_m8 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
diff --git a/libgfortran/generated/matmul_l16.c b/libgfortran/generated/matmul_l16.c
index 3eb886907918..a210a6b470e8 100644
--- a/libgfortran/generated/matmul_l16.c
+++ b/libgfortran/generated/matmul_l16.c
@@ -161,13 +161,13 @@ matmul_l16 (gfc_array_l16 * const restrict retarray,
 
   if (GFC_DESCRIPTOR_RANK (retarray) == 1)
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rxstride = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride = rxstride;
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
-      rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
+      rxstride = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
     }
 
   /* If we have rank 1 parameters, zero the absent stride, and set the size to
diff --git a/libgfortran/generated/matmul_l4.c b/libgfortran/generated/matmul_l4.c
index 76144e5ee83e..2c294b8d0c98 100644
--- a/libgfortran/generated/matmul_l4.c
+++ b/libgfortran/generated/matmul_l4.c
@@ -161,13 +161,13 @@ matmul_l4 (gfc_array_l4 * const restrict retarray,
 
   if (GFC_DESCRIPTOR_RANK (retarray) == 1)
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rxstride = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride = rxstride;
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
-      rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
+      rxstride = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
     }
 
   /* If we have rank 1 parameters, zero the absent stride, and set the size to
diff --git a/libgfortran/generated/matmul_l8.c b/libgfortran/generated/matmul_l8.c
index 9f22e4a48134..63ee634fdc8f 100644
--- a/libgfortran/generated/matmul_l8.c
+++ b/libgfortran/generated/matmul_l8.c
@@ -161,13 +161,13 @@ matmul_l8 (gfc_array_l8 * const restrict retarray,
 
   if (GFC_DESCRIPTOR_RANK (retarray) == 1)
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rxstride = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride = rxstride;
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
-      rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
+      rxstride = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
     }
 
   /* If we have rank 1 parameters, zero the absent stride, and set the size to
diff --git a/libgfortran/generated/matmul_r10.c b/libgfortran/generated/matmul_r10.c
index 5f3cc97cbf21..b16c0a9aa16e 100644
--- a/libgfortran/generated/matmul_r10.c
+++ b/libgfortran/generated/matmul_r10.c
@@ -92,7 +92,7 @@ matmul_r10_avx (gfc_array_r10 * const restrict retarray,
   const GFC_REAL_10 * restrict bbase;
   GFC_REAL_10 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -183,12 +183,11 @@ matmul_r10_avx (gfc_array_r10 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -257,15 +256,19 @@ matmul_r10_avx (gfc_array_r10 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_REAL_10)
+      && (axstride_bytes == sizeof (GFC_REAL_10)
+	  || aystride_bytes == sizeof (GFC_REAL_10))
+      && (bxstride_bytes == sizeof (GFC_REAL_10)
+	  || bystride_bytes == sizeof (GFC_REAL_10))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_REAL_10 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_REAL_10)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_REAL_10)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -274,12 +277,12 @@ matmul_r10_avx (gfc_array_r10 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_REAL_10) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_REAL_10) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -288,7 +291,9 @@ matmul_r10_avx (gfc_array_r10 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_REAL_10)
+      && axstride_bytes == sizeof (GFC_REAL_10)
+      && bxstride_bytes == sizeof (GFC_REAL_10)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -341,7 +346,7 @@ matmul_r10_avx (gfc_array_r10 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_REAL_10))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -567,7 +572,9 @@ matmul_r10_avx (gfc_array_r10 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_REAL_10)
+	   && aystride_bytes == sizeof (GFC_REAL_10)
+	   && bxstride_bytes == sizeof (GFC_REAL_10))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -620,7 +627,7 @@ matmul_r10_avx (gfc_array_r10 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -677,7 +684,7 @@ matmul_r10_avx2 (gfc_array_r10 * const restrict retarray,
   const GFC_REAL_10 * restrict bbase;
   GFC_REAL_10 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -768,12 +775,11 @@ matmul_r10_avx2 (gfc_array_r10 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -842,15 +848,19 @@ matmul_r10_avx2 (gfc_array_r10 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_REAL_10)
+      && (axstride_bytes == sizeof (GFC_REAL_10)
+	  || aystride_bytes == sizeof (GFC_REAL_10))
+      && (bxstride_bytes == sizeof (GFC_REAL_10)
+	  || bystride_bytes == sizeof (GFC_REAL_10))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_REAL_10 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_REAL_10)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_REAL_10)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -859,12 +869,12 @@ matmul_r10_avx2 (gfc_array_r10 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_REAL_10) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_REAL_10) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -873,7 +883,9 @@ matmul_r10_avx2 (gfc_array_r10 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_REAL_10)
+      && axstride_bytes == sizeof (GFC_REAL_10)
+      && bxstride_bytes == sizeof (GFC_REAL_10)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -926,7 +938,7 @@ matmul_r10_avx2 (gfc_array_r10 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_REAL_10))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -1152,7 +1164,9 @@ matmul_r10_avx2 (gfc_array_r10 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_REAL_10)
+	   && aystride_bytes == sizeof (GFC_REAL_10)
+	   && bxstride_bytes == sizeof (GFC_REAL_10))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -1205,7 +1219,7 @@ matmul_r10_avx2 (gfc_array_r10 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -1262,7 +1276,7 @@ matmul_r10_avx512f (gfc_array_r10 * const restrict retarray,
   const GFC_REAL_10 * restrict bbase;
   GFC_REAL_10 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -1353,12 +1367,11 @@ matmul_r10_avx512f (gfc_array_r10 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -1427,15 +1440,19 @@ matmul_r10_avx512f (gfc_array_r10 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_REAL_10)
+      && (axstride_bytes == sizeof (GFC_REAL_10)
+	  || aystride_bytes == sizeof (GFC_REAL_10))
+      && (bxstride_bytes == sizeof (GFC_REAL_10)
+	  || bystride_bytes == sizeof (GFC_REAL_10))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_REAL_10 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_REAL_10)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_REAL_10)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -1444,12 +1461,12 @@ matmul_r10_avx512f (gfc_array_r10 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_REAL_10) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_REAL_10) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -1458,7 +1475,9 @@ matmul_r10_avx512f (gfc_array_r10 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_REAL_10)
+      && axstride_bytes == sizeof (GFC_REAL_10)
+      && bxstride_bytes == sizeof (GFC_REAL_10)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -1511,7 +1530,7 @@ matmul_r10_avx512f (gfc_array_r10 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_REAL_10))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -1737,7 +1756,9 @@ matmul_r10_avx512f (gfc_array_r10 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_REAL_10)
+	   && aystride_bytes == sizeof (GFC_REAL_10)
+	   && bxstride_bytes == sizeof (GFC_REAL_10))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -1790,7 +1811,7 @@ matmul_r10_avx512f (gfc_array_r10 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -1861,7 +1882,7 @@ matmul_r10_vanilla (gfc_array_r10 * const restrict retarray,
   const GFC_REAL_10 * restrict bbase;
   GFC_REAL_10 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -1952,12 +1973,11 @@ matmul_r10_vanilla (gfc_array_r10 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -2026,15 +2046,19 @@ matmul_r10_vanilla (gfc_array_r10 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_REAL_10)
+      && (axstride_bytes == sizeof (GFC_REAL_10)
+	  || aystride_bytes == sizeof (GFC_REAL_10))
+      && (bxstride_bytes == sizeof (GFC_REAL_10)
+	  || bystride_bytes == sizeof (GFC_REAL_10))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_REAL_10 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_REAL_10)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_REAL_10)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -2043,12 +2067,12 @@ matmul_r10_vanilla (gfc_array_r10 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_REAL_10) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_REAL_10) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -2057,7 +2081,9 @@ matmul_r10_vanilla (gfc_array_r10 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_REAL_10)
+      && axstride_bytes == sizeof (GFC_REAL_10)
+      && bxstride_bytes == sizeof (GFC_REAL_10)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -2110,7 +2136,7 @@ matmul_r10_vanilla (gfc_array_r10 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_REAL_10))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -2336,7 +2362,9 @@ matmul_r10_vanilla (gfc_array_r10 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_REAL_10)
+	   && aystride_bytes == sizeof (GFC_REAL_10)
+	   && bxstride_bytes == sizeof (GFC_REAL_10))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -2389,7 +2417,7 @@ matmul_r10_vanilla (gfc_array_r10 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -2519,7 +2547,7 @@ matmul_r10 (gfc_array_r10 * const restrict retarray,
   const GFC_REAL_10 * restrict bbase;
   GFC_REAL_10 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -2610,12 +2638,11 @@ matmul_r10 (gfc_array_r10 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -2684,15 +2711,19 @@ matmul_r10 (gfc_array_r10 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_REAL_10)
+      && (axstride_bytes == sizeof (GFC_REAL_10)
+	  || aystride_bytes == sizeof (GFC_REAL_10))
+      && (bxstride_bytes == sizeof (GFC_REAL_10)
+	  || bystride_bytes == sizeof (GFC_REAL_10))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_REAL_10 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_REAL_10)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_REAL_10)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -2701,12 +2732,12 @@ matmul_r10 (gfc_array_r10 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_REAL_10) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_REAL_10) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -2715,7 +2746,9 @@ matmul_r10 (gfc_array_r10 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_REAL_10)
+      && axstride_bytes == sizeof (GFC_REAL_10)
+      && bxstride_bytes == sizeof (GFC_REAL_10)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -2768,7 +2801,7 @@ matmul_r10 (gfc_array_r10 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_REAL_10))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -2994,7 +3027,9 @@ matmul_r10 (gfc_array_r10 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_REAL_10)
+	   && aystride_bytes == sizeof (GFC_REAL_10)
+	   && bxstride_bytes == sizeof (GFC_REAL_10))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -3047,7 +3082,7 @@ matmul_r10 (gfc_array_r10 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
diff --git a/libgfortran/generated/matmul_r16.c b/libgfortran/generated/matmul_r16.c
index 43d7cf4d5140..b58d4d6f28c7 100644
--- a/libgfortran/generated/matmul_r16.c
+++ b/libgfortran/generated/matmul_r16.c
@@ -92,7 +92,7 @@ matmul_r16_avx (gfc_array_r16 * const restrict retarray,
   const GFC_REAL_16 * restrict bbase;
   GFC_REAL_16 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -183,12 +183,11 @@ matmul_r16_avx (gfc_array_r16 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -257,15 +256,19 @@ matmul_r16_avx (gfc_array_r16 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_REAL_16)
+      && (axstride_bytes == sizeof (GFC_REAL_16)
+	  || aystride_bytes == sizeof (GFC_REAL_16))
+      && (bxstride_bytes == sizeof (GFC_REAL_16)
+	  || bystride_bytes == sizeof (GFC_REAL_16))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_REAL_16 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_REAL_16)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_REAL_16)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -274,12 +277,12 @@ matmul_r16_avx (gfc_array_r16 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_REAL_16) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_REAL_16) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -288,7 +291,9 @@ matmul_r16_avx (gfc_array_r16 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_REAL_16)
+      && axstride_bytes == sizeof (GFC_REAL_16)
+      && bxstride_bytes == sizeof (GFC_REAL_16)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -341,7 +346,7 @@ matmul_r16_avx (gfc_array_r16 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_REAL_16))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -567,7 +572,9 @@ matmul_r16_avx (gfc_array_r16 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_REAL_16)
+	   && aystride_bytes == sizeof (GFC_REAL_16)
+	   && bxstride_bytes == sizeof (GFC_REAL_16))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -620,7 +627,7 @@ matmul_r16_avx (gfc_array_r16 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -677,7 +684,7 @@ matmul_r16_avx2 (gfc_array_r16 * const restrict retarray,
   const GFC_REAL_16 * restrict bbase;
   GFC_REAL_16 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -768,12 +775,11 @@ matmul_r16_avx2 (gfc_array_r16 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -842,15 +848,19 @@ matmul_r16_avx2 (gfc_array_r16 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_REAL_16)
+      && (axstride_bytes == sizeof (GFC_REAL_16)
+	  || aystride_bytes == sizeof (GFC_REAL_16))
+      && (bxstride_bytes == sizeof (GFC_REAL_16)
+	  || bystride_bytes == sizeof (GFC_REAL_16))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_REAL_16 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_REAL_16)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_REAL_16)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -859,12 +869,12 @@ matmul_r16_avx2 (gfc_array_r16 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_REAL_16) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_REAL_16) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -873,7 +883,9 @@ matmul_r16_avx2 (gfc_array_r16 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_REAL_16)
+      && axstride_bytes == sizeof (GFC_REAL_16)
+      && bxstride_bytes == sizeof (GFC_REAL_16)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -926,7 +938,7 @@ matmul_r16_avx2 (gfc_array_r16 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_REAL_16))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -1152,7 +1164,9 @@ matmul_r16_avx2 (gfc_array_r16 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_REAL_16)
+	   && aystride_bytes == sizeof (GFC_REAL_16)
+	   && bxstride_bytes == sizeof (GFC_REAL_16))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -1205,7 +1219,7 @@ matmul_r16_avx2 (gfc_array_r16 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -1262,7 +1276,7 @@ matmul_r16_avx512f (gfc_array_r16 * const restrict retarray,
   const GFC_REAL_16 * restrict bbase;
   GFC_REAL_16 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -1353,12 +1367,11 @@ matmul_r16_avx512f (gfc_array_r16 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -1427,15 +1440,19 @@ matmul_r16_avx512f (gfc_array_r16 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_REAL_16)
+      && (axstride_bytes == sizeof (GFC_REAL_16)
+	  || aystride_bytes == sizeof (GFC_REAL_16))
+      && (bxstride_bytes == sizeof (GFC_REAL_16)
+	  || bystride_bytes == sizeof (GFC_REAL_16))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_REAL_16 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_REAL_16)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_REAL_16)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -1444,12 +1461,12 @@ matmul_r16_avx512f (gfc_array_r16 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_REAL_16) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_REAL_16) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -1458,7 +1475,9 @@ matmul_r16_avx512f (gfc_array_r16 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_REAL_16)
+      && axstride_bytes == sizeof (GFC_REAL_16)
+      && bxstride_bytes == sizeof (GFC_REAL_16)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -1511,7 +1530,7 @@ matmul_r16_avx512f (gfc_array_r16 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_REAL_16))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -1737,7 +1756,9 @@ matmul_r16_avx512f (gfc_array_r16 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_REAL_16)
+	   && aystride_bytes == sizeof (GFC_REAL_16)
+	   && bxstride_bytes == sizeof (GFC_REAL_16))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -1790,7 +1811,7 @@ matmul_r16_avx512f (gfc_array_r16 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -1861,7 +1882,7 @@ matmul_r16_vanilla (gfc_array_r16 * const restrict retarray,
   const GFC_REAL_16 * restrict bbase;
   GFC_REAL_16 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -1952,12 +1973,11 @@ matmul_r16_vanilla (gfc_array_r16 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -2026,15 +2046,19 @@ matmul_r16_vanilla (gfc_array_r16 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_REAL_16)
+      && (axstride_bytes == sizeof (GFC_REAL_16)
+	  || aystride_bytes == sizeof (GFC_REAL_16))
+      && (bxstride_bytes == sizeof (GFC_REAL_16)
+	  || bystride_bytes == sizeof (GFC_REAL_16))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_REAL_16 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_REAL_16)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_REAL_16)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -2043,12 +2067,12 @@ matmul_r16_vanilla (gfc_array_r16 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_REAL_16) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_REAL_16) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -2057,7 +2081,9 @@ matmul_r16_vanilla (gfc_array_r16 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_REAL_16)
+      && axstride_bytes == sizeof (GFC_REAL_16)
+      && bxstride_bytes == sizeof (GFC_REAL_16)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -2110,7 +2136,7 @@ matmul_r16_vanilla (gfc_array_r16 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_REAL_16))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -2336,7 +2362,9 @@ matmul_r16_vanilla (gfc_array_r16 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_REAL_16)
+	   && aystride_bytes == sizeof (GFC_REAL_16)
+	   && bxstride_bytes == sizeof (GFC_REAL_16))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -2389,7 +2417,7 @@ matmul_r16_vanilla (gfc_array_r16 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -2519,7 +2547,7 @@ matmul_r16 (gfc_array_r16 * const restrict retarray,
   const GFC_REAL_16 * restrict bbase;
   GFC_REAL_16 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -2610,12 +2638,11 @@ matmul_r16 (gfc_array_r16 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -2684,15 +2711,19 @@ matmul_r16 (gfc_array_r16 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_REAL_16)
+      && (axstride_bytes == sizeof (GFC_REAL_16)
+	  || aystride_bytes == sizeof (GFC_REAL_16))
+      && (bxstride_bytes == sizeof (GFC_REAL_16)
+	  || bystride_bytes == sizeof (GFC_REAL_16))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_REAL_16 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_REAL_16)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_REAL_16)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -2701,12 +2732,12 @@ matmul_r16 (gfc_array_r16 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_REAL_16) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_REAL_16) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -2715,7 +2746,9 @@ matmul_r16 (gfc_array_r16 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_REAL_16)
+      && axstride_bytes == sizeof (GFC_REAL_16)
+      && bxstride_bytes == sizeof (GFC_REAL_16)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -2768,7 +2801,7 @@ matmul_r16 (gfc_array_r16 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_REAL_16))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -2994,7 +3027,9 @@ matmul_r16 (gfc_array_r16 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_REAL_16)
+	   && aystride_bytes == sizeof (GFC_REAL_16)
+	   && bxstride_bytes == sizeof (GFC_REAL_16))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -3047,7 +3082,7 @@ matmul_r16 (gfc_array_r16 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
diff --git a/libgfortran/generated/matmul_r17.c b/libgfortran/generated/matmul_r17.c
index dc08ed149ef4..3012bd00664e 100644
--- a/libgfortran/generated/matmul_r17.c
+++ b/libgfortran/generated/matmul_r17.c
@@ -92,7 +92,7 @@ matmul_r17_avx (gfc_array_r17 * const restrict retarray,
   const GFC_REAL_17 * restrict bbase;
   GFC_REAL_17 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -183,12 +183,11 @@ matmul_r17_avx (gfc_array_r17 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -257,15 +256,19 @@ matmul_r17_avx (gfc_array_r17 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_REAL_17)
+      && (axstride_bytes == sizeof (GFC_REAL_17)
+	  || aystride_bytes == sizeof (GFC_REAL_17))
+      && (bxstride_bytes == sizeof (GFC_REAL_17)
+	  || bystride_bytes == sizeof (GFC_REAL_17))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_REAL_17 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_REAL_17)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_REAL_17)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -274,12 +277,12 @@ matmul_r17_avx (gfc_array_r17 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_REAL_17) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_REAL_17) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -288,7 +291,9 @@ matmul_r17_avx (gfc_array_r17 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_REAL_17)
+      && axstride_bytes == sizeof (GFC_REAL_17)
+      && bxstride_bytes == sizeof (GFC_REAL_17)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -341,7 +346,7 @@ matmul_r17_avx (gfc_array_r17 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_REAL_17))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -567,7 +572,9 @@ matmul_r17_avx (gfc_array_r17 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_REAL_17)
+	   && aystride_bytes == sizeof (GFC_REAL_17)
+	   && bxstride_bytes == sizeof (GFC_REAL_17))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -620,7 +627,7 @@ matmul_r17_avx (gfc_array_r17 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -677,7 +684,7 @@ matmul_r17_avx2 (gfc_array_r17 * const restrict retarray,
   const GFC_REAL_17 * restrict bbase;
   GFC_REAL_17 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -768,12 +775,11 @@ matmul_r17_avx2 (gfc_array_r17 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -842,15 +848,19 @@ matmul_r17_avx2 (gfc_array_r17 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_REAL_17)
+      && (axstride_bytes == sizeof (GFC_REAL_17)
+	  || aystride_bytes == sizeof (GFC_REAL_17))
+      && (bxstride_bytes == sizeof (GFC_REAL_17)
+	  || bystride_bytes == sizeof (GFC_REAL_17))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_REAL_17 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_REAL_17)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_REAL_17)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -859,12 +869,12 @@ matmul_r17_avx2 (gfc_array_r17 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_REAL_17) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_REAL_17) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -873,7 +883,9 @@ matmul_r17_avx2 (gfc_array_r17 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_REAL_17)
+      && axstride_bytes == sizeof (GFC_REAL_17)
+      && bxstride_bytes == sizeof (GFC_REAL_17)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -926,7 +938,7 @@ matmul_r17_avx2 (gfc_array_r17 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_REAL_17))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -1152,7 +1164,9 @@ matmul_r17_avx2 (gfc_array_r17 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_REAL_17)
+	   && aystride_bytes == sizeof (GFC_REAL_17)
+	   && bxstride_bytes == sizeof (GFC_REAL_17))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -1205,7 +1219,7 @@ matmul_r17_avx2 (gfc_array_r17 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -1262,7 +1276,7 @@ matmul_r17_avx512f (gfc_array_r17 * const restrict retarray,
   const GFC_REAL_17 * restrict bbase;
   GFC_REAL_17 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -1353,12 +1367,11 @@ matmul_r17_avx512f (gfc_array_r17 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -1427,15 +1440,19 @@ matmul_r17_avx512f (gfc_array_r17 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_REAL_17)
+      && (axstride_bytes == sizeof (GFC_REAL_17)
+	  || aystride_bytes == sizeof (GFC_REAL_17))
+      && (bxstride_bytes == sizeof (GFC_REAL_17)
+	  || bystride_bytes == sizeof (GFC_REAL_17))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_REAL_17 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_REAL_17)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_REAL_17)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -1444,12 +1461,12 @@ matmul_r17_avx512f (gfc_array_r17 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_REAL_17) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_REAL_17) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -1458,7 +1475,9 @@ matmul_r17_avx512f (gfc_array_r17 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_REAL_17)
+      && axstride_bytes == sizeof (GFC_REAL_17)
+      && bxstride_bytes == sizeof (GFC_REAL_17)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -1511,7 +1530,7 @@ matmul_r17_avx512f (gfc_array_r17 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_REAL_17))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -1737,7 +1756,9 @@ matmul_r17_avx512f (gfc_array_r17 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_REAL_17)
+	   && aystride_bytes == sizeof (GFC_REAL_17)
+	   && bxstride_bytes == sizeof (GFC_REAL_17))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -1790,7 +1811,7 @@ matmul_r17_avx512f (gfc_array_r17 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -1861,7 +1882,7 @@ matmul_r17_vanilla (gfc_array_r17 * const restrict retarray,
   const GFC_REAL_17 * restrict bbase;
   GFC_REAL_17 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -1952,12 +1973,11 @@ matmul_r17_vanilla (gfc_array_r17 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -2026,15 +2046,19 @@ matmul_r17_vanilla (gfc_array_r17 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_REAL_17)
+      && (axstride_bytes == sizeof (GFC_REAL_17)
+	  || aystride_bytes == sizeof (GFC_REAL_17))
+      && (bxstride_bytes == sizeof (GFC_REAL_17)
+	  || bystride_bytes == sizeof (GFC_REAL_17))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_REAL_17 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_REAL_17)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_REAL_17)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -2043,12 +2067,12 @@ matmul_r17_vanilla (gfc_array_r17 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_REAL_17) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_REAL_17) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -2057,7 +2081,9 @@ matmul_r17_vanilla (gfc_array_r17 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_REAL_17)
+      && axstride_bytes == sizeof (GFC_REAL_17)
+      && bxstride_bytes == sizeof (GFC_REAL_17)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -2110,7 +2136,7 @@ matmul_r17_vanilla (gfc_array_r17 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_REAL_17))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -2336,7 +2362,9 @@ matmul_r17_vanilla (gfc_array_r17 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_REAL_17)
+	   && aystride_bytes == sizeof (GFC_REAL_17)
+	   && bxstride_bytes == sizeof (GFC_REAL_17))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -2389,7 +2417,7 @@ matmul_r17_vanilla (gfc_array_r17 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -2519,7 +2547,7 @@ matmul_r17 (gfc_array_r17 * const restrict retarray,
   const GFC_REAL_17 * restrict bbase;
   GFC_REAL_17 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -2610,12 +2638,11 @@ matmul_r17 (gfc_array_r17 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -2684,15 +2711,19 @@ matmul_r17 (gfc_array_r17 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_REAL_17)
+      && (axstride_bytes == sizeof (GFC_REAL_17)
+	  || aystride_bytes == sizeof (GFC_REAL_17))
+      && (bxstride_bytes == sizeof (GFC_REAL_17)
+	  || bystride_bytes == sizeof (GFC_REAL_17))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_REAL_17 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_REAL_17)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_REAL_17)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -2701,12 +2732,12 @@ matmul_r17 (gfc_array_r17 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_REAL_17) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_REAL_17) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -2715,7 +2746,9 @@ matmul_r17 (gfc_array_r17 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_REAL_17)
+      && axstride_bytes == sizeof (GFC_REAL_17)
+      && bxstride_bytes == sizeof (GFC_REAL_17)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -2768,7 +2801,7 @@ matmul_r17 (gfc_array_r17 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_REAL_17))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -2994,7 +3027,9 @@ matmul_r17 (gfc_array_r17 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_REAL_17)
+	   && aystride_bytes == sizeof (GFC_REAL_17)
+	   && bxstride_bytes == sizeof (GFC_REAL_17))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -3047,7 +3082,7 @@ matmul_r17 (gfc_array_r17 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
diff --git a/libgfortran/generated/matmul_r4.c b/libgfortran/generated/matmul_r4.c
index ee36b1ec6d0d..8fcdbe835f6c 100644
--- a/libgfortran/generated/matmul_r4.c
+++ b/libgfortran/generated/matmul_r4.c
@@ -92,7 +92,7 @@ matmul_r4_avx (gfc_array_r4 * const restrict retarray,
   const GFC_REAL_4 * restrict bbase;
   GFC_REAL_4 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -183,12 +183,11 @@ matmul_r4_avx (gfc_array_r4 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -257,15 +256,19 @@ matmul_r4_avx (gfc_array_r4 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_REAL_4)
+      && (axstride_bytes == sizeof (GFC_REAL_4)
+	  || aystride_bytes == sizeof (GFC_REAL_4))
+      && (bxstride_bytes == sizeof (GFC_REAL_4)
+	  || bystride_bytes == sizeof (GFC_REAL_4))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_REAL_4 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_REAL_4)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_REAL_4)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -274,12 +277,12 @@ matmul_r4_avx (gfc_array_r4 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_REAL_4) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_REAL_4) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -288,7 +291,9 @@ matmul_r4_avx (gfc_array_r4 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_REAL_4)
+      && axstride_bytes == sizeof (GFC_REAL_4)
+      && bxstride_bytes == sizeof (GFC_REAL_4)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -341,7 +346,7 @@ matmul_r4_avx (gfc_array_r4 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_REAL_4))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -567,7 +572,9 @@ matmul_r4_avx (gfc_array_r4 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_REAL_4)
+	   && aystride_bytes == sizeof (GFC_REAL_4)
+	   && bxstride_bytes == sizeof (GFC_REAL_4))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -620,7 +627,7 @@ matmul_r4_avx (gfc_array_r4 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -677,7 +684,7 @@ matmul_r4_avx2 (gfc_array_r4 * const restrict retarray,
   const GFC_REAL_4 * restrict bbase;
   GFC_REAL_4 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -768,12 +775,11 @@ matmul_r4_avx2 (gfc_array_r4 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -842,15 +848,19 @@ matmul_r4_avx2 (gfc_array_r4 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_REAL_4)
+      && (axstride_bytes == sizeof (GFC_REAL_4)
+	  || aystride_bytes == sizeof (GFC_REAL_4))
+      && (bxstride_bytes == sizeof (GFC_REAL_4)
+	  || bystride_bytes == sizeof (GFC_REAL_4))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_REAL_4 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_REAL_4)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_REAL_4)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -859,12 +869,12 @@ matmul_r4_avx2 (gfc_array_r4 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_REAL_4) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_REAL_4) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -873,7 +883,9 @@ matmul_r4_avx2 (gfc_array_r4 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_REAL_4)
+      && axstride_bytes == sizeof (GFC_REAL_4)
+      && bxstride_bytes == sizeof (GFC_REAL_4)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -926,7 +938,7 @@ matmul_r4_avx2 (gfc_array_r4 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_REAL_4))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -1152,7 +1164,9 @@ matmul_r4_avx2 (gfc_array_r4 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_REAL_4)
+	   && aystride_bytes == sizeof (GFC_REAL_4)
+	   && bxstride_bytes == sizeof (GFC_REAL_4))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -1205,7 +1219,7 @@ matmul_r4_avx2 (gfc_array_r4 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -1262,7 +1276,7 @@ matmul_r4_avx512f (gfc_array_r4 * const restrict retarray,
   const GFC_REAL_4 * restrict bbase;
   GFC_REAL_4 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -1353,12 +1367,11 @@ matmul_r4_avx512f (gfc_array_r4 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -1427,15 +1440,19 @@ matmul_r4_avx512f (gfc_array_r4 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_REAL_4)
+      && (axstride_bytes == sizeof (GFC_REAL_4)
+	  || aystride_bytes == sizeof (GFC_REAL_4))
+      && (bxstride_bytes == sizeof (GFC_REAL_4)
+	  || bystride_bytes == sizeof (GFC_REAL_4))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_REAL_4 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_REAL_4)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_REAL_4)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -1444,12 +1461,12 @@ matmul_r4_avx512f (gfc_array_r4 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_REAL_4) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_REAL_4) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -1458,7 +1475,9 @@ matmul_r4_avx512f (gfc_array_r4 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_REAL_4)
+      && axstride_bytes == sizeof (GFC_REAL_4)
+      && bxstride_bytes == sizeof (GFC_REAL_4)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -1511,7 +1530,7 @@ matmul_r4_avx512f (gfc_array_r4 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_REAL_4))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -1737,7 +1756,9 @@ matmul_r4_avx512f (gfc_array_r4 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_REAL_4)
+	   && aystride_bytes == sizeof (GFC_REAL_4)
+	   && bxstride_bytes == sizeof (GFC_REAL_4))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -1790,7 +1811,7 @@ matmul_r4_avx512f (gfc_array_r4 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -1861,7 +1882,7 @@ matmul_r4_vanilla (gfc_array_r4 * const restrict retarray,
   const GFC_REAL_4 * restrict bbase;
   GFC_REAL_4 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -1952,12 +1973,11 @@ matmul_r4_vanilla (gfc_array_r4 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -2026,15 +2046,19 @@ matmul_r4_vanilla (gfc_array_r4 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_REAL_4)
+      && (axstride_bytes == sizeof (GFC_REAL_4)
+	  || aystride_bytes == sizeof (GFC_REAL_4))
+      && (bxstride_bytes == sizeof (GFC_REAL_4)
+	  || bystride_bytes == sizeof (GFC_REAL_4))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_REAL_4 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_REAL_4)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_REAL_4)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -2043,12 +2067,12 @@ matmul_r4_vanilla (gfc_array_r4 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_REAL_4) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_REAL_4) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -2057,7 +2081,9 @@ matmul_r4_vanilla (gfc_array_r4 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_REAL_4)
+      && axstride_bytes == sizeof (GFC_REAL_4)
+      && bxstride_bytes == sizeof (GFC_REAL_4)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -2110,7 +2136,7 @@ matmul_r4_vanilla (gfc_array_r4 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_REAL_4))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -2336,7 +2362,9 @@ matmul_r4_vanilla (gfc_array_r4 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_REAL_4)
+	   && aystride_bytes == sizeof (GFC_REAL_4)
+	   && bxstride_bytes == sizeof (GFC_REAL_4))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -2389,7 +2417,7 @@ matmul_r4_vanilla (gfc_array_r4 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -2519,7 +2547,7 @@ matmul_r4 (gfc_array_r4 * const restrict retarray,
   const GFC_REAL_4 * restrict bbase;
   GFC_REAL_4 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -2610,12 +2638,11 @@ matmul_r4 (gfc_array_r4 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -2684,15 +2711,19 @@ matmul_r4 (gfc_array_r4 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_REAL_4)
+      && (axstride_bytes == sizeof (GFC_REAL_4)
+	  || aystride_bytes == sizeof (GFC_REAL_4))
+      && (bxstride_bytes == sizeof (GFC_REAL_4)
+	  || bystride_bytes == sizeof (GFC_REAL_4))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_REAL_4 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_REAL_4)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_REAL_4)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -2701,12 +2732,12 @@ matmul_r4 (gfc_array_r4 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_REAL_4) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_REAL_4) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -2715,7 +2746,9 @@ matmul_r4 (gfc_array_r4 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_REAL_4)
+      && axstride_bytes == sizeof (GFC_REAL_4)
+      && bxstride_bytes == sizeof (GFC_REAL_4)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -2768,7 +2801,7 @@ matmul_r4 (gfc_array_r4 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_REAL_4))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -2994,7 +3027,9 @@ matmul_r4 (gfc_array_r4 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_REAL_4)
+	   && aystride_bytes == sizeof (GFC_REAL_4)
+	   && bxstride_bytes == sizeof (GFC_REAL_4))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -3047,7 +3082,7 @@ matmul_r4 (gfc_array_r4 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
diff --git a/libgfortran/generated/matmul_r8.c b/libgfortran/generated/matmul_r8.c
index c67dcc9b7617..8a380faab732 100644
--- a/libgfortran/generated/matmul_r8.c
+++ b/libgfortran/generated/matmul_r8.c
@@ -92,7 +92,7 @@ matmul_r8_avx (gfc_array_r8 * const restrict retarray,
   const GFC_REAL_8 * restrict bbase;
   GFC_REAL_8 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -183,12 +183,11 @@ matmul_r8_avx (gfc_array_r8 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -257,15 +256,19 @@ matmul_r8_avx (gfc_array_r8 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_REAL_8)
+      && (axstride_bytes == sizeof (GFC_REAL_8)
+	  || aystride_bytes == sizeof (GFC_REAL_8))
+      && (bxstride_bytes == sizeof (GFC_REAL_8)
+	  || bystride_bytes == sizeof (GFC_REAL_8))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_REAL_8 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_REAL_8)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_REAL_8)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -274,12 +277,12 @@ matmul_r8_avx (gfc_array_r8 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_REAL_8) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_REAL_8) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -288,7 +291,9 @@ matmul_r8_avx (gfc_array_r8 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_REAL_8)
+      && axstride_bytes == sizeof (GFC_REAL_8)
+      && bxstride_bytes == sizeof (GFC_REAL_8)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -341,7 +346,7 @@ matmul_r8_avx (gfc_array_r8 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_REAL_8))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -567,7 +572,9 @@ matmul_r8_avx (gfc_array_r8 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_REAL_8)
+	   && aystride_bytes == sizeof (GFC_REAL_8)
+	   && bxstride_bytes == sizeof (GFC_REAL_8))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -620,7 +627,7 @@ matmul_r8_avx (gfc_array_r8 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -677,7 +684,7 @@ matmul_r8_avx2 (gfc_array_r8 * const restrict retarray,
   const GFC_REAL_8 * restrict bbase;
   GFC_REAL_8 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -768,12 +775,11 @@ matmul_r8_avx2 (gfc_array_r8 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -842,15 +848,19 @@ matmul_r8_avx2 (gfc_array_r8 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_REAL_8)
+      && (axstride_bytes == sizeof (GFC_REAL_8)
+	  || aystride_bytes == sizeof (GFC_REAL_8))
+      && (bxstride_bytes == sizeof (GFC_REAL_8)
+	  || bystride_bytes == sizeof (GFC_REAL_8))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_REAL_8 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_REAL_8)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_REAL_8)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -859,12 +869,12 @@ matmul_r8_avx2 (gfc_array_r8 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_REAL_8) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_REAL_8) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -873,7 +883,9 @@ matmul_r8_avx2 (gfc_array_r8 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_REAL_8)
+      && axstride_bytes == sizeof (GFC_REAL_8)
+      && bxstride_bytes == sizeof (GFC_REAL_8)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -926,7 +938,7 @@ matmul_r8_avx2 (gfc_array_r8 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_REAL_8))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -1152,7 +1164,9 @@ matmul_r8_avx2 (gfc_array_r8 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_REAL_8)
+	   && aystride_bytes == sizeof (GFC_REAL_8)
+	   && bxstride_bytes == sizeof (GFC_REAL_8))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -1205,7 +1219,7 @@ matmul_r8_avx2 (gfc_array_r8 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -1262,7 +1276,7 @@ matmul_r8_avx512f (gfc_array_r8 * const restrict retarray,
   const GFC_REAL_8 * restrict bbase;
   GFC_REAL_8 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -1353,12 +1367,11 @@ matmul_r8_avx512f (gfc_array_r8 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -1427,15 +1440,19 @@ matmul_r8_avx512f (gfc_array_r8 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_REAL_8)
+      && (axstride_bytes == sizeof (GFC_REAL_8)
+	  || aystride_bytes == sizeof (GFC_REAL_8))
+      && (bxstride_bytes == sizeof (GFC_REAL_8)
+	  || bystride_bytes == sizeof (GFC_REAL_8))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_REAL_8 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_REAL_8)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_REAL_8)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -1444,12 +1461,12 @@ matmul_r8_avx512f (gfc_array_r8 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_REAL_8) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_REAL_8) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -1458,7 +1475,9 @@ matmul_r8_avx512f (gfc_array_r8 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_REAL_8)
+      && axstride_bytes == sizeof (GFC_REAL_8)
+      && bxstride_bytes == sizeof (GFC_REAL_8)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -1511,7 +1530,7 @@ matmul_r8_avx512f (gfc_array_r8 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_REAL_8))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -1737,7 +1756,9 @@ matmul_r8_avx512f (gfc_array_r8 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_REAL_8)
+	   && aystride_bytes == sizeof (GFC_REAL_8)
+	   && bxstride_bytes == sizeof (GFC_REAL_8))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -1790,7 +1811,7 @@ matmul_r8_avx512f (gfc_array_r8 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -1861,7 +1882,7 @@ matmul_r8_vanilla (gfc_array_r8 * const restrict retarray,
   const GFC_REAL_8 * restrict bbase;
   GFC_REAL_8 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -1952,12 +1973,11 @@ matmul_r8_vanilla (gfc_array_r8 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -2026,15 +2046,19 @@ matmul_r8_vanilla (gfc_array_r8 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_REAL_8)
+      && (axstride_bytes == sizeof (GFC_REAL_8)
+	  || aystride_bytes == sizeof (GFC_REAL_8))
+      && (bxstride_bytes == sizeof (GFC_REAL_8)
+	  || bystride_bytes == sizeof (GFC_REAL_8))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_REAL_8 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_REAL_8)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_REAL_8)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -2043,12 +2067,12 @@ matmul_r8_vanilla (gfc_array_r8 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_REAL_8) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_REAL_8) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -2057,7 +2081,9 @@ matmul_r8_vanilla (gfc_array_r8 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_REAL_8)
+      && axstride_bytes == sizeof (GFC_REAL_8)
+      && bxstride_bytes == sizeof (GFC_REAL_8)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -2110,7 +2136,7 @@ matmul_r8_vanilla (gfc_array_r8 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_REAL_8))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -2336,7 +2362,9 @@ matmul_r8_vanilla (gfc_array_r8 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_REAL_8)
+	   && aystride_bytes == sizeof (GFC_REAL_8)
+	   && bxstride_bytes == sizeof (GFC_REAL_8))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -2389,7 +2417,7 @@ matmul_r8_vanilla (gfc_array_r8 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -2519,7 +2547,7 @@ matmul_r8 (gfc_array_r8 * const restrict retarray,
   const GFC_REAL_8 * restrict bbase;
   GFC_REAL_8 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -2610,12 +2638,11 @@ matmul_r8 (gfc_array_r8 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -2684,15 +2711,19 @@ matmul_r8 (gfc_array_r8 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_REAL_8)
+      && (axstride_bytes == sizeof (GFC_REAL_8)
+	  || aystride_bytes == sizeof (GFC_REAL_8))
+      && (bxstride_bytes == sizeof (GFC_REAL_8)
+	  || bystride_bytes == sizeof (GFC_REAL_8))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_REAL_8 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_REAL_8)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_REAL_8)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -2701,12 +2732,12 @@ matmul_r8 (gfc_array_r8 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_REAL_8) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_REAL_8) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -2715,7 +2746,9 @@ matmul_r8 (gfc_array_r8 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_REAL_8)
+      && axstride_bytes == sizeof (GFC_REAL_8)
+      && bxstride_bytes == sizeof (GFC_REAL_8)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -2768,7 +2801,7 @@ matmul_r8 (gfc_array_r8 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_REAL_8))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -2994,7 +3027,9 @@ matmul_r8 (gfc_array_r8 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_REAL_8)
+	   && aystride_bytes == sizeof (GFC_REAL_8)
+	   && bxstride_bytes == sizeof (GFC_REAL_8))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -3047,7 +3082,7 @@ matmul_r8 (gfc_array_r8 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
diff --git a/libgfortran/generated/matmulavx128_c10.c b/libgfortran/generated/matmulavx128_c10.c
index 3df3a43a59bb..5b4b5a47ceea 100644
--- a/libgfortran/generated/matmulavx128_c10.c
+++ b/libgfortran/generated/matmulavx128_c10.c
@@ -57,7 +57,7 @@ matmul_c10_avx128_fma3 (gfc_array_c10 * const restrict retarray,
   const GFC_COMPLEX_10 * restrict bbase;
   GFC_COMPLEX_10 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -148,12 +148,11 @@ matmul_c10_avx128_fma3 (gfc_array_c10 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -222,15 +221,19 @@ matmul_c10_avx128_fma3 (gfc_array_c10 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_COMPLEX_10)
+      && (axstride_bytes == sizeof (GFC_COMPLEX_10)
+	  || aystride_bytes == sizeof (GFC_COMPLEX_10))
+      && (bxstride_bytes == sizeof (GFC_COMPLEX_10)
+	  || bystride_bytes == sizeof (GFC_COMPLEX_10))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_COMPLEX_10 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_COMPLEX_10)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_COMPLEX_10)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -239,12 +242,12 @@ matmul_c10_avx128_fma3 (gfc_array_c10 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_COMPLEX_10) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_COMPLEX_10) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -253,7 +256,9 @@ matmul_c10_avx128_fma3 (gfc_array_c10 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_COMPLEX_10)
+      && axstride_bytes == sizeof (GFC_COMPLEX_10)
+      && bxstride_bytes == sizeof (GFC_COMPLEX_10)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -306,7 +311,7 @@ matmul_c10_avx128_fma3 (gfc_array_c10 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_COMPLEX_10))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -532,7 +537,9 @@ matmul_c10_avx128_fma3 (gfc_array_c10 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_COMPLEX_10)
+	   && aystride_bytes == sizeof (GFC_COMPLEX_10)
+	   && bxstride_bytes == sizeof (GFC_COMPLEX_10))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -585,7 +592,7 @@ matmul_c10_avx128_fma3 (gfc_array_c10 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -643,7 +650,7 @@ matmul_c10_avx128_fma4 (gfc_array_c10 * const restrict retarray,
   const GFC_COMPLEX_10 * restrict bbase;
   GFC_COMPLEX_10 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -734,12 +741,11 @@ matmul_c10_avx128_fma4 (gfc_array_c10 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -808,15 +814,19 @@ matmul_c10_avx128_fma4 (gfc_array_c10 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_COMPLEX_10)
+      && (axstride_bytes == sizeof (GFC_COMPLEX_10)
+	  || aystride_bytes == sizeof (GFC_COMPLEX_10))
+      && (bxstride_bytes == sizeof (GFC_COMPLEX_10)
+	  || bystride_bytes == sizeof (GFC_COMPLEX_10))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_COMPLEX_10 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_COMPLEX_10)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_COMPLEX_10)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -825,12 +835,12 @@ matmul_c10_avx128_fma4 (gfc_array_c10 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_COMPLEX_10) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_COMPLEX_10) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -839,7 +849,9 @@ matmul_c10_avx128_fma4 (gfc_array_c10 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_COMPLEX_10)
+      && axstride_bytes == sizeof (GFC_COMPLEX_10)
+      && bxstride_bytes == sizeof (GFC_COMPLEX_10)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -892,7 +904,7 @@ matmul_c10_avx128_fma4 (gfc_array_c10 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_COMPLEX_10))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -1118,7 +1130,9 @@ matmul_c10_avx128_fma4 (gfc_array_c10 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_COMPLEX_10)
+	   && aystride_bytes == sizeof (GFC_COMPLEX_10)
+	   && bxstride_bytes == sizeof (GFC_COMPLEX_10))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -1171,7 +1185,7 @@ matmul_c10_avx128_fma4 (gfc_array_c10 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
diff --git a/libgfortran/generated/matmulavx128_c16.c b/libgfortran/generated/matmulavx128_c16.c
index 11263fa2d3d3..31804fad9f43 100644
--- a/libgfortran/generated/matmulavx128_c16.c
+++ b/libgfortran/generated/matmulavx128_c16.c
@@ -57,7 +57,7 @@ matmul_c16_avx128_fma3 (gfc_array_c16 * const restrict retarray,
   const GFC_COMPLEX_16 * restrict bbase;
   GFC_COMPLEX_16 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -148,12 +148,11 @@ matmul_c16_avx128_fma3 (gfc_array_c16 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -222,15 +221,19 @@ matmul_c16_avx128_fma3 (gfc_array_c16 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_COMPLEX_16)
+      && (axstride_bytes == sizeof (GFC_COMPLEX_16)
+	  || aystride_bytes == sizeof (GFC_COMPLEX_16))
+      && (bxstride_bytes == sizeof (GFC_COMPLEX_16)
+	  || bystride_bytes == sizeof (GFC_COMPLEX_16))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_COMPLEX_16 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_COMPLEX_16)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_COMPLEX_16)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -239,12 +242,12 @@ matmul_c16_avx128_fma3 (gfc_array_c16 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_COMPLEX_16) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_COMPLEX_16) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -253,7 +256,9 @@ matmul_c16_avx128_fma3 (gfc_array_c16 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_COMPLEX_16)
+      && axstride_bytes == sizeof (GFC_COMPLEX_16)
+      && bxstride_bytes == sizeof (GFC_COMPLEX_16)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -306,7 +311,7 @@ matmul_c16_avx128_fma3 (gfc_array_c16 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_COMPLEX_16))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -532,7 +537,9 @@ matmul_c16_avx128_fma3 (gfc_array_c16 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_COMPLEX_16)
+	   && aystride_bytes == sizeof (GFC_COMPLEX_16)
+	   && bxstride_bytes == sizeof (GFC_COMPLEX_16))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -585,7 +592,7 @@ matmul_c16_avx128_fma3 (gfc_array_c16 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -643,7 +650,7 @@ matmul_c16_avx128_fma4 (gfc_array_c16 * const restrict retarray,
   const GFC_COMPLEX_16 * restrict bbase;
   GFC_COMPLEX_16 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -734,12 +741,11 @@ matmul_c16_avx128_fma4 (gfc_array_c16 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -808,15 +814,19 @@ matmul_c16_avx128_fma4 (gfc_array_c16 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_COMPLEX_16)
+      && (axstride_bytes == sizeof (GFC_COMPLEX_16)
+	  || aystride_bytes == sizeof (GFC_COMPLEX_16))
+      && (bxstride_bytes == sizeof (GFC_COMPLEX_16)
+	  || bystride_bytes == sizeof (GFC_COMPLEX_16))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_COMPLEX_16 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_COMPLEX_16)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_COMPLEX_16)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -825,12 +835,12 @@ matmul_c16_avx128_fma4 (gfc_array_c16 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_COMPLEX_16) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_COMPLEX_16) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -839,7 +849,9 @@ matmul_c16_avx128_fma4 (gfc_array_c16 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_COMPLEX_16)
+      && axstride_bytes == sizeof (GFC_COMPLEX_16)
+      && bxstride_bytes == sizeof (GFC_COMPLEX_16)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -892,7 +904,7 @@ matmul_c16_avx128_fma4 (gfc_array_c16 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_COMPLEX_16))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -1118,7 +1130,9 @@ matmul_c16_avx128_fma4 (gfc_array_c16 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_COMPLEX_16)
+	   && aystride_bytes == sizeof (GFC_COMPLEX_16)
+	   && bxstride_bytes == sizeof (GFC_COMPLEX_16))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -1171,7 +1185,7 @@ matmul_c16_avx128_fma4 (gfc_array_c16 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
diff --git a/libgfortran/generated/matmulavx128_c17.c b/libgfortran/generated/matmulavx128_c17.c
index 8a6957200827..bee088707672 100644
--- a/libgfortran/generated/matmulavx128_c17.c
+++ b/libgfortran/generated/matmulavx128_c17.c
@@ -57,7 +57,7 @@ matmul_c17_avx128_fma3 (gfc_array_c17 * const restrict retarray,
   const GFC_COMPLEX_17 * restrict bbase;
   GFC_COMPLEX_17 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -148,12 +148,11 @@ matmul_c17_avx128_fma3 (gfc_array_c17 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -222,15 +221,19 @@ matmul_c17_avx128_fma3 (gfc_array_c17 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_COMPLEX_17)
+      && (axstride_bytes == sizeof (GFC_COMPLEX_17)
+	  || aystride_bytes == sizeof (GFC_COMPLEX_17))
+      && (bxstride_bytes == sizeof (GFC_COMPLEX_17)
+	  || bystride_bytes == sizeof (GFC_COMPLEX_17))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_COMPLEX_17 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_COMPLEX_17)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_COMPLEX_17)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -239,12 +242,12 @@ matmul_c17_avx128_fma3 (gfc_array_c17 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_COMPLEX_17) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_COMPLEX_17) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -253,7 +256,9 @@ matmul_c17_avx128_fma3 (gfc_array_c17 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_COMPLEX_17)
+      && axstride_bytes == sizeof (GFC_COMPLEX_17)
+      && bxstride_bytes == sizeof (GFC_COMPLEX_17)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -306,7 +311,7 @@ matmul_c17_avx128_fma3 (gfc_array_c17 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_COMPLEX_17))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -532,7 +537,9 @@ matmul_c17_avx128_fma3 (gfc_array_c17 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_COMPLEX_17)
+	   && aystride_bytes == sizeof (GFC_COMPLEX_17)
+	   && bxstride_bytes == sizeof (GFC_COMPLEX_17))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -585,7 +592,7 @@ matmul_c17_avx128_fma3 (gfc_array_c17 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -643,7 +650,7 @@ matmul_c17_avx128_fma4 (gfc_array_c17 * const restrict retarray,
   const GFC_COMPLEX_17 * restrict bbase;
   GFC_COMPLEX_17 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -734,12 +741,11 @@ matmul_c17_avx128_fma4 (gfc_array_c17 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -808,15 +814,19 @@ matmul_c17_avx128_fma4 (gfc_array_c17 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_COMPLEX_17)
+      && (axstride_bytes == sizeof (GFC_COMPLEX_17)
+	  || aystride_bytes == sizeof (GFC_COMPLEX_17))
+      && (bxstride_bytes == sizeof (GFC_COMPLEX_17)
+	  || bystride_bytes == sizeof (GFC_COMPLEX_17))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_COMPLEX_17 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_COMPLEX_17)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_COMPLEX_17)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -825,12 +835,12 @@ matmul_c17_avx128_fma4 (gfc_array_c17 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_COMPLEX_17) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_COMPLEX_17) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -839,7 +849,9 @@ matmul_c17_avx128_fma4 (gfc_array_c17 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_COMPLEX_17)
+      && axstride_bytes == sizeof (GFC_COMPLEX_17)
+      && bxstride_bytes == sizeof (GFC_COMPLEX_17)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -892,7 +904,7 @@ matmul_c17_avx128_fma4 (gfc_array_c17 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_COMPLEX_17))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -1118,7 +1130,9 @@ matmul_c17_avx128_fma4 (gfc_array_c17 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_COMPLEX_17)
+	   && aystride_bytes == sizeof (GFC_COMPLEX_17)
+	   && bxstride_bytes == sizeof (GFC_COMPLEX_17))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -1171,7 +1185,7 @@ matmul_c17_avx128_fma4 (gfc_array_c17 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
diff --git a/libgfortran/generated/matmulavx128_c4.c b/libgfortran/generated/matmulavx128_c4.c
index 2d47cc7f6041..c27a4e976db6 100644
--- a/libgfortran/generated/matmulavx128_c4.c
+++ b/libgfortran/generated/matmulavx128_c4.c
@@ -57,7 +57,7 @@ matmul_c4_avx128_fma3 (gfc_array_c4 * const restrict retarray,
   const GFC_COMPLEX_4 * restrict bbase;
   GFC_COMPLEX_4 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -148,12 +148,11 @@ matmul_c4_avx128_fma3 (gfc_array_c4 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -222,15 +221,19 @@ matmul_c4_avx128_fma3 (gfc_array_c4 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_COMPLEX_4)
+      && (axstride_bytes == sizeof (GFC_COMPLEX_4)
+	  || aystride_bytes == sizeof (GFC_COMPLEX_4))
+      && (bxstride_bytes == sizeof (GFC_COMPLEX_4)
+	  || bystride_bytes == sizeof (GFC_COMPLEX_4))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_COMPLEX_4 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_COMPLEX_4)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_COMPLEX_4)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -239,12 +242,12 @@ matmul_c4_avx128_fma3 (gfc_array_c4 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_COMPLEX_4) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_COMPLEX_4) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -253,7 +256,9 @@ matmul_c4_avx128_fma3 (gfc_array_c4 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_COMPLEX_4)
+      && axstride_bytes == sizeof (GFC_COMPLEX_4)
+      && bxstride_bytes == sizeof (GFC_COMPLEX_4)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -306,7 +311,7 @@ matmul_c4_avx128_fma3 (gfc_array_c4 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_COMPLEX_4))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -532,7 +537,9 @@ matmul_c4_avx128_fma3 (gfc_array_c4 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_COMPLEX_4)
+	   && aystride_bytes == sizeof (GFC_COMPLEX_4)
+	   && bxstride_bytes == sizeof (GFC_COMPLEX_4))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -585,7 +592,7 @@ matmul_c4_avx128_fma3 (gfc_array_c4 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
@@ -643,7 +650,7 @@ matmul_c4_avx128_fma4 (gfc_array_c4 * const restrict retarray,
   const GFC_COMPLEX_4 * restrict bbase;
   GFC_COMPLEX_4 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -734,12 +741,11 @@ matmul_c4_avx128_fma4 (gfc_array_c4 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -808,15 +814,19 @@ matmul_c4_avx128_fma4 (gfc_array_c4 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_COMPLEX_4)
+      && (axstride_bytes == sizeof (GFC_COMPLEX_4)
+	  || aystride_bytes == sizeof (GFC_COMPLEX_4))
+      && (bxstride_bytes == sizeof (GFC_COMPLEX_4)
+	  || bystride_bytes == sizeof (GFC_COMPLEX_4))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_COMPLEX_4 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_COMPLEX_4)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_COMPLEX_4)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -825,12 +835,12 @@ matmul_c4_avx128_fma4 (gfc_array_c4 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_COMPLEX_4) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_COMPLEX_4) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -839,7 +849,9 @@ matmul_c4_avx128_fma4 (gfc_array_c4 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_COMPLEX_4)
+      && axstride_bytes == sizeof (GFC_COMPLEX_4)
+      && bxstride_bytes == sizeof (GFC_COMPLEX_4)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -892,7 +904,7 @@ matmul_c4_avx128_fma4 (gfc_array_c4 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_COMPLEX_4))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -1118,7 +1130,9 @@ matmul_c4_avx128_fma4 (gfc_array_c4 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_COMPLEX_4)
+	   && aystride_bytes == sizeof (GFC_COMPLEX_4)
+	   && bxstride_bytes == sizeof (GFC_COMPLEX_4))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -1171,7 +1185,7 @@ matmul_c4_avx128_fma4 (gfc_array_c4 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes < aystride_bytes)
     {
       for (y = 0; y < ycount; y++)
 	for (x = 0; x < xcount; x++)
diff --git a/libgfortran/generated/matmulavx128_c8.c b/libgfortran/generated/matmulavx128_c8.c
index 61c19042e8b5..fe7194e28946 100644
--- a/libgfortran/generated/matmulavx128_c8.c
+++ b/libgfortran/generated/matmulavx128_c8.c
@@ -57,7 +57,7 @@ matmul_c8_avx128_fma3 (gfc_array_c8 * const restrict retarray,
   const GFC_COMPLEX_8 * restrict bbase;
   GFC_COMPLEX_8 * restrict dest;
 
-  index_type rxstride, rystride, axstride, aystride, bxstride, bystride;
+  index_type rystride, axstride, aystride, bxstride, bystride;
   index_type x, y, n, count, xcount, ycount;
   index_type axstride_bytes, aystride_bytes, bxstride_bytes, bystride_bytes,
 	     rxstride_bytes, rystride_bytes;
@@ -148,12 +148,11 @@ matmul_c8_avx128_fma3 (gfc_array_c8 * const restrict retarray,
       /* One-dimensional result may be addressed in the code below
 	 either as a row or a column matrix. We want both cases to
 	 work. */
-      rxstride = rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
+      rystride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rxstride_bytes = rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
     }
   else
     {
-      rxstride = GFC_DESCRIPTOR_STRIDE(retarray,0);
       rystride = GFC_DESCRIPTOR_STRIDE(retarray,1);
       rxstride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,0);
       rystride_bytes = GFC_DESCRIPTOR_STRIDE_BYTES(retarray,1);
@@ -222,15 +221,19 @@ matmul_c8_avx128_fma3 (gfc_array_c8 * const restrict retarray,
 #define min(a,b) ((a) <= (b) ? (a) : (b))
 #define max(a,b) ((a) >= (b) ? (a) : (b))
 
-  if (try_blas && rxstride == 1 && (axstride == 1 || aystride == 1)
-      && (bxstride == 1 || bystride == 1)
+  if (try_blas
+      && rxstride_bytes == sizeof (GFC_COMPLEX_8)
+      && (axstride_bytes == sizeof (GFC_COMPLEX_8)
+	  || aystride_bytes == sizeof (GFC_COMPLEX_8))
+      && (bxstride_bytes == sizeof (GFC_COMPLEX_8)
+	  || bystride_bytes == sizeof (GFC_COMPLEX_8))
       && (((float) xcount) * ((float) ycount) * ((float) count)
           > POW3(blas_limit)))
     {
       const int m = xcount, n = ycount, k = count, ldc = rystride;
       const GFC_COMPLEX_8 one = 1, zero = 0;
-      const int lda = (axstride == 1) ? aystride : axstride,
-		ldb = (bxstride == 1) ? bystride : bxstride;
+      const int lda = (axstride_bytes == sizeof (GFC_COMPLEX_8)) ? aystride : axstride,
+		ldb = (bxstride_bytes == sizeof (GFC_COMPLEX_8)) ? bystride : bxstride;
 
       if (lda > 0 && ldb > 0 && ldc > 0 && m > 1 && n > 1 && k > 1)
 	{
@@ -239,12 +242,12 @@ matmul_c8_avx128_fma3 (gfc_array_c8 * const restrict retarray,
 	  if (try_blas & 2)
 	    transa = "C";
 	  else
-	    transa = axstride == 1 ? "N" : "T";
+	    transa = axstride_bytes == sizeof (GFC_COMPLEX_8) ? "N" : "T";
 
 	  if (try_blas & 4)
 	    transb = "C";
 	  else
-	    transb = bxstride == 1 ? "N" : "T";
+	    transb = bxstride_bytes == sizeof (GFC_COMPLEX_8) ? "N" : "T";
 
 	  gemm (transa, transb , &m,
 		&n, &k,	&one, abase, &lda, bbase, &ldb, &zero, dest,
@@ -253,7 +256,9 @@ matmul_c8_avx128_fma3 (gfc_array_c8 * const restrict retarray,
 	}
     }
 
-  if (rxstride == 1 && axstride == 1 && bxstride == 1
+  if (rxstride_bytes == sizeof (GFC_COMPLEX_8)
+      && axstride_bytes == sizeof (GFC_COMPLEX_8)
+      && bxstride_bytes == sizeof (GFC_COMPLEX_8)
       && GFC_DESCRIPTOR_RANK (b) != 1)
     {
       /* This block of code implements a tuned matmul, derived from
@@ -306,7 +311,7 @@ matmul_c8_avx128_fma3 (gfc_array_c8 * const restrict retarray,
 
       /* Adjust size of t1 to what is needed.  */
       index_type t1_dim, a_sz;
-      if (aystride == 1)
+      if (aystride_bytes == sizeof (GFC_COMPLEX_8))
         a_sz = rystride;
       else
         a_sz = a_dim1;
@@ -532,7 +537,9 @@ matmul_c8_avx128_fma3 (gfc_array_c8 * const restrict retarray,
 #undef B_ARRAY_ELEM
 #undef C_ARRAY_ELEM
     }
-  else if (rxstride == 1 && aystride == 1 && bxstride == 1)
+  else if (rxstride_bytes == sizeof (GFC_COMPLEX_8)
+	   && aystride_bytes == sizeof (GFC_COMPLEX_8)
+	   && bxstride_bytes == sizeof (GFC_COMPLEX_8))
     {
       if (GFC_DESCRIPTOR_RANK (a) != 1)
 	{
@@ -585,7 +592,7 @@ matmul_c8_avx128_fma3 (gfc_array_c8 * const restrict retarray,
 	  GFC_DESCRIPTOR1_ELEM (retarray, y) = s;
 	}
     }
-  else if (axstride < aystride)
+  else if (axstride_bytes [...]

[diff truncated at 524288 bytes]


More information about the Gcc-cvs mailing list