g++ optimizer and numerical computations
Marc Espie
Marc.Espie@liafa.jussieu.fr
Mon Feb 16 09:06:00 GMT 1998
I'm trying to design a refcnted vector type for numerical computations,
and benching that on the 980205 egcs snapshot with haifa-enabled on an
alpha. If I'm not mistaken in my analysis, the results make me less
than happy.
Code fragment:
#include <iostream>
#include <stdlib.h>
const unsigned int SIZE=6349; // (0)
class InVector
{
size_t cnt;
int data[SIZE];
friend class Vector;
InVector(): cnt(1) {};
};
class Vector
{
InVector *value;
public:
const int operator [](size_t i) const
{
return value->data[i];
}
int &operator [](size_t i)
{
if (value->cnt != 1)
// value = value.copy();
{
InVector *v = new(InVector);
for (size_t j = 0; j < SIZE; j++)
v->data[j] = value->data[j];
value->cnt--;
value = v;
}
value->cnt = 1;
return value->data[i];
}
Vector() {value = new(InVector);};
~Vector()
{
if (value->cnt-- == 1)
delete value;
}
Vector(const Vector &v)
{
value = v.value;
value->cnt++;
}
};
int main()
{
Vector v;
for (size_t i = 0; i < SIZE; i++)
v[i] = i;
Vector w = v; // (1)
for (size_t i = 0; i < SIZE; i++)
w[i] = 5; // (2)
for (size_t i = 0; i < SIZE; i++)
cout << v[i] << " ";
cout << endl;
for (size_t i = 0; i < SIZE; i++)
cout << w[i] << " ";
cout << endl;
}
The SIZE (0) insures the compiler won't just unroll the whole loop.
The two interesting points are labelled (1) and (2).
(1) is the vector copy, which just Does a cnt++ on the InVector value.
(2) implements copy on write semantics: w[i] is used as an lvalue, hence
the int &Vector::operator [](size_t i) must be selected, and if cnt > 1,
the copy does occur.
All operators are defined inline to make it possible for the compiler to
notice that modifications to cnt don't occur elsewhere.
With that code, g++ -O9 spouts the following assembler output, where
the code we want occurs between $L515 and $L516.
.verstamp 3 11
.set noreorder
.set volatile
.set noat
.file 1 "test.C"
gcc2_compiled.:
__gnu_compiled_cplusplus:
.rdata
.quad 0
$LC0:
.ascii " \0"
.text
.align 3
.globl main
.ent main
main:
ldgp $29,0($27)
$main..ng:
lda $30,-224($30)
.frame $30,224,$26,0
stq $26,0($30)
stq $9,8($30)
stq $10,16($30)
stq $11,24($30)
stq $12,32($30)
stq $13,40($30)
stq $14,48($30)
stq $15,56($30)
.mask 0x400fe00,-224
.prologue 1
jsr $26,__get_eh_context
ldgp $29,0($26)
lda $16,25408
bis $0,$0,$13
addq $30,72,$10
addq $30,64,$11
jsr $26,__builtin_new
ldgp $29,0($26)
bis $13,$13,$9
ldt $f1,0($13)
bis $13,$13,$12
lda $1,$L489
bis $13,$13,$2
stq $1,96($30)
stt $f1,72($30)
stq $31,80($30)
stq $10,88($30)
stq $30,104($30)
br $31,$L490
.align 3
$L489:
$LSJ35:
ldgp $29,$LSJ35-$L489($27)
br $31,$L488
.align 3
$L490:
stq $10,0($2)
bis $31,1,$1
stq $1,0($0)
ldq $1,0($9)
ldt $f1,0($1)
stt $f1,0($9)
stq $0,0($11)
br $31,$L494
.align 3
$L488:
bis $0,$0,$16
jsr $26,__builtin_delete
ldgp $29,0($26)
jsr $26,__sjthrow
ldgp $29,0($26)
.align 3
$L494:
ldq $2,0($13)
addq $30,64,$6
lda $1,_$_6Vector
addq $30,120,$3
bis $31,$31,$11
ldt $f1,8($2)
stq $1,128($30)
stq $6,136($30)
stt $f1,120($30)
stq $3,8($2)
.align 3
$L496:
lda $1,6348
cmpule $11,$1,$1
beq $1,$L497
ldq $1,64($30)
bis $12,$12,$9
addq $30,72,$10
addq $30,64,$15
ldq $1,0($1)
cmpeq $1,1,$1
bne $1,$L500
lda $16,25408
jsr $26,__builtin_new
ldgp $29,0($26)
ldt $f1,0($12)
bis $12,$12,$2
lda $1,$L502
stq $1,96($30)
stt $f1,72($30)
stq $31,80($30)
stq $10,88($30)
stq $30,104($30)
br $31,$L503
.align 3
$L502:
$LSJ109:
ldgp $29,$LSJ109-$L502($27)
br $31,$L501
.align 3
$L503:
stq $10,0($2)
bis $31,1,$1
stq $1,0($0)
bis $31,$31,$2
ldq $1,0($9)
lda $4,6348
addq $0,8,$3
ldt $f1,0($1)
stt $f1,0($9)
.align 3
$L508:
ldq $1,0($15)
s4addq $2,$1,$1
addq $2,1,$2
lds $f1,8($1)
cmpule $2,$4,$1
sts $f1,0($3)
addq $3,4,$3
bne $1,$L508
ldq $2,0($15)
ldq $1,0($2)
subq $1,1,$1
stq $1,0($2)
stq $0,0($15)
$L500:
ldq $2,0($15)
bis $31,1,$1
s4addq $11,8,$3
stq $1,0($2)
ldq $1,0($15)
addq $1,$3,$1
br $31,$L511
.align 3
$L501:
bis $0,$0,$16
jsr $26,__builtin_delete
ldgp $29,0($26)
jsr $26,__sjthrow
ldgp $29,0($26)
.align 3
$L511:
stl $11,0($1)
addq $11,1,$11
br $31,$L496
.align 3
$L497:
ldq $2,64($30)
addq $30,72,$4
lda $3,_$_6Vector
addq $30,144,$5
bis $31,$31,$10
stq $2,72($30)
ldq $1,0($2)
addq $1,1,$1
stq $1,0($2)
ldq $1,0($13)
ldt $f1,8($1)
stq $3,152($30)
stq $4,160($30)
stt $f1,144($30)
stq $5,8($1)
.align 3
$L515:
lda $1,6348
cmpule $10,$1,$1
beq $1,$L516
ldq $1,72($30)
bis $12,$12,$15
addq $30,168,$9
ldq $1,0($1)
cmpeq $1,1,$1
bne $1,$L519
lda $16,25408
jsr $26,__builtin_new
ldgp $29,0($26)
ldt $f1,0($12)
bis $12,$12,$2
lda $1,$L521
stq $1,192($30)
stt $f1,168($30)
stq $31,176($30)
stq $9,184($30)
stq $30,200($30)
br $31,$L522
.align 3
$L521:
$LSJ240:
ldgp $29,$LSJ240-$L521($27)
br $31,$L520
.align 3
$L522:
stq $9,0($2)
bis $31,1,$1
stq $1,0($0)
bis $31,$31,$2
ldq $1,0($15)
lda $4,6348
addq $0,8,$3
ldt $f1,0($1)
stt $f1,0($15)
.align 3
$L527:
ldq $1,72($30)
s4addq $2,$1,$1
addq $2,1,$2
lds $f1,8($1)
cmpule $2,$4,$1
sts $f1,0($3)
addq $3,4,$3
bne $1,$L527
ldq $2,72($30)
ldq $1,0($2)
subq $1,1,$1
stq $1,0($2)
stq $0,72($30)
$L519:
ldq $2,72($30)
bis $31,1,$1
s4addq $10,8,$3
stq $1,0($2)
ldq $1,72($30)
addq $1,$3,$2
br $31,$L530
.align 3
$L520:
bis $0,$0,$16
jsr $26,__builtin_delete
ldgp $29,0($26)
jsr $26,__sjthrow
ldgp $29,0($26)
.align 3
$L530:
bis $31,5,$1
stl $1,0($2)
addq $10,1,$10
br $31,$L515
.align 3
$L516:
bis $31,$31,$11
.align 3
$L532:
lda $1,6348
cmpule $11,$1,$1
beq $1,$L533
ldq $1,64($30)
bis $12,$12,$9
lda $14,cout
addq $30,168,$10
ldq $1,0($1)
addq $30,64,$15
cmpeq $1,1,$1
bne $1,$L536
lda $16,25408
jsr $26,__builtin_new
ldgp $29,0($26)
ldt $f1,0($12)
bis $12,$12,$2
lda $1,$L538
stq $1,192($30)
stt $f1,168($30)
stq $31,176($30)
stq $10,184($30)
stq $30,200($30)
br $31,$L539
.align 3
$L538:
$LSJ355:
ldgp $29,$LSJ355-$L538($27)
br $31,$L537
.align 3
$L539:
stq $10,0($2)
bis $31,1,$1
stq $1,0($0)
bis $31,$31,$2
ldq $1,0($9)
lda $4,6348
addq $0,8,$3
ldt $f1,0($1)
stt $f1,0($9)
.align 3
$L544:
ldq $1,0($15)
s4addq $2,$1,$1
addq $2,1,$2
lds $f1,8($1)
cmpule $2,$4,$1
sts $f1,0($3)
addq $3,4,$3
bne $1,$L544
ldq $2,0($15)
ldq $1,0($2)
subq $1,1,$1
stq $1,0($2)
stq $0,0($15)
$L536:
ldq $2,0($15)
bis $31,1,$1
s4addq $11,8,$3
stq $1,0($2)
ldq $1,0($15)
addq $1,$3,$1
br $31,$L547
.align 3
$L537:
bis $0,$0,$16
jsr $26,__builtin_delete
ldgp $29,0($26)
jsr $26,__sjthrow
ldgp $29,0($26)
.align 3
$L547:
ldl $17,0($1)
bis $14,$14,$16
addq $11,1,$11
jsr $26,__ls__7ostreami
ldgp $29,0($26)
lda $17,$LC0
bis $0,$0,$16
jsr $26,__ls__7ostreamPCc
ldgp $29,0($26)
br $31,$L532
.align 3
$L533:
lda $16,cout
bis $31,$31,$10
lda $27,endl__FR7ostream
jsr $26,($27),0
ldgp $29,0($26)
.align 3
$L550:
lda $1,6348
cmpule $10,$1,$1
beq $1,$L551
ldq $1,72($30)
bis $12,$12,$15
lda $11,cout
addq $30,168,$9
ldq $1,0($1)
cmpeq $1,1,$1
bne $1,$L554
lda $16,25408
jsr $26,__builtin_new
ldgp $29,0($26)
ldt $f1,0($15)
bis $15,$15,$2
lda $1,$L556
stq $1,192($30)
stt $f1,168($30)
stq $31,176($30)
stq $9,184($30)
stq $30,200($30)
br $31,$L557
.align 3
$L556:
$LSJ484:
ldgp $29,$LSJ484-$L556($27)
br $31,$L555
.align 3
$L557:
stq $9,0($2)
bis $31,1,$1
stq $1,0($0)
bis $31,$31,$2
ldq $1,0($15)
lda $4,6348
addq $0,8,$3
ldt $f1,0($1)
stt $f1,0($15)
.align 3
$L562:
ldq $1,72($30)
s4addq $2,$1,$1
addq $2,1,$2
lds $f1,8($1)
cmpule $2,$4,$1
sts $f1,0($3)
addq $3,4,$3
bne $1,$L562
ldq $2,72($30)
ldq $1,0($2)
subq $1,1,$1
stq $1,0($2)
stq $0,72($30)
$L554:
ldq $2,72($30)
bis $31,1,$1
s4addq $10,8,$3
stq $1,0($2)
ldq $1,72($30)
addq $1,$3,$1
br $31,$L565
.align 3
$L555:
bis $0,$0,$16
jsr $26,__builtin_delete
ldgp $29,0($26)
jsr $26,__sjthrow
ldgp $29,0($26)
.align 3
$L565:
ldl $17,0($1)
bis $11,$11,$16
addq $10,1,$10
jsr $26,__ls__7ostreami
ldgp $29,0($26)
lda $17,$LC0
bis $0,$0,$16
jsr $26,__ls__7ostreamPCc
ldgp $29,0($26)
br $31,$L550
.align 3
$L551:
lda $16,cout
lda $27,endl__FR7ostream
jsr $26,($27),0
ldgp $29,0($26)
ldq $2,0($13)
ldq $1,8($2)
ldt $f1,0($1)
stt $f1,8($2)
ldq $2,72($30)
ldq $1,0($2)
subq $1,1,$1
stq $1,0($2)
bne $1,$L572
ldq $16,72($30)
jsr $26,__builtin_delete
ldgp $29,0($26)
$L572:
ldq $2,0($13)
ldq $1,8($2)
ldt $f1,0($1)
stt $f1,8($2)
ldq $2,64($30)
ldq $1,0($2)
subq $1,1,$1
stq $1,0($2)
bne $1,$L577
ldq $16,64($30)
jsr $26,__builtin_delete
ldgp $29,0($26)
$L577:
bis $31,$31,$0
ldq $26,0($30)
ldq $9,8($30)
ldq $10,16($30)
ldq $11,24($30)
ldq $12,32($30)
ldq $13,40($30)
ldq $14,48($30)
ldq $15,56($30)
addq $30,224,$30
ret $31,($26),1
.end main
.align 3
.ent _$_6Vector
_$_6Vector:
ldgp $29,0($27)
$_$_6Vector..ng:
lda $30,-32($30)
.frame $30,32,$26,0
stq $26,0($30)
stq $9,8($30)
stq $10,16($30)
.mask 0x4000600,-32
.prologue 1
bis $16,$16,$9
bis $17,$17,$10
ldq $2,0($9)
ldq $1,0($2)
subq $1,1,$1
stq $1,0($2)
bne $1,$L482
ldq $16,0($9)
jsr $26,__builtin_delete
ldgp $29,0($26)
$L482:
bis $9,$9,$16
and $10,1,$1
beq $1,$L484
jsr $26,__builtin_delete
ldgp $29,0($26)
$L484:
ldq $26,0($30)
ldq $9,8($30)
ldq $10,16($30)
addq $30,32,$30
ret $31,($26),1
.end _$_6Vector
unless I'm mistaken, the loop test is
lda $1, 6348
cmpule $10,$1, $1
beq $1, $L516
and the refcnt test is
ldq $1,0($1)
cmpeq $1,1,$1
bne $1,$L519
- the refcnt test occurs inside the loop. It has not been moved outside
the loop by the compiler.
- the exceptional case (refcnt != 1) is inside the loop, and the normal
case (refcnt == 1) is coded as a forward conditional branch, predicted
to fall through (alpha architecture handbook).
Did I miss something, or is the optimization truly as bad as it looks ?
Any optimization options I missed ? Any way to code the loop so that it
will run faster ?
--
Marc Espie
More information about the Gcc
mailing list