g++ optimizer and numerical computations

Marc Espie Marc.Espie@liafa.jussieu.fr
Mon Feb 16 09:06:00 GMT 1998


I'm trying to design a refcnted vector type for numerical computations, 
and benching that on the 980205 egcs snapshot with haifa-enabled on an
alpha.  If I'm not mistaken in my analysis, the results make me less
than happy.

Code fragment:

#include <iostream>
#include <stdlib.h>

const unsigned int SIZE=6349; // (0)
class InVector
	{
	size_t cnt;
	int data[SIZE];
	friend class Vector;
	InVector(): cnt(1) {};
	};

class Vector
	{

	InVector *value;

public:
	const int operator [](size_t i) const
		{
		return value->data[i];
		}
	int &operator [](size_t i)
		{
		if (value->cnt != 1)
//			value = value.copy();
			{
			InVector *v = new(InVector);
			for (size_t j = 0; j < SIZE; j++)
				v->data[j] = value->data[j];
			value->cnt--;
			value = v;
			}
		value->cnt = 1;
		return value->data[i];
		}
	Vector() {value = new(InVector);};
	~Vector() 
		{ 
		if (value->cnt-- == 1) 
			delete value;
		}
	Vector(const Vector &v)
		{
		value = v.value;
		value->cnt++;
		}
	};

int main()
	{
	Vector v;
	for (size_t i = 0; i < SIZE; i++)
		v[i] = i;

	Vector w = v;						// (1)
	for (size_t i = 0; i < SIZE; i++)
		w[i] = 5;						// (2)

	for (size_t i = 0; i < SIZE; i++)
		cout << v[i] << " ";
	cout << endl;
	for (size_t i = 0; i < SIZE; i++)
		cout << w[i] << " ";
	cout << endl;
	}


The SIZE (0) insures the compiler won't just unroll the whole loop.
The two interesting points are labelled (1) and (2).
(1) is the vector copy, which just Does a cnt++ on the InVector value.
(2) implements copy on write semantics: w[i] is used as an lvalue, hence
the int &Vector::operator [](size_t i) must be selected, and if cnt > 1,
the copy does occur.

All operators are defined inline to make it possible for the compiler to
notice that modifications to cnt don't occur elsewhere.

With that code, g++ -O9 spouts the following assembler output, where
the code we want occurs between $L515 and $L516.

	.verstamp 3 11
	.set noreorder
	.set volatile
	.set noat
	.file	1 "test.C"
gcc2_compiled.:
__gnu_compiled_cplusplus:
.rdata
	.quad 0
$LC0:
	.ascii " \0"
.text
	.align 3
	.globl main
	.ent main
main:
	ldgp $29,0($27)
$main..ng:
	lda $30,-224($30)
	.frame $30,224,$26,0
	stq $26,0($30)
	stq $9,8($30)
	stq $10,16($30)
	stq $11,24($30)
	stq $12,32($30)
	stq $13,40($30)
	stq $14,48($30)
	stq $15,56($30)
	.mask 0x400fe00,-224
	.prologue 1
	jsr $26,__get_eh_context
	ldgp $29,0($26)
	lda $16,25408
	bis $0,$0,$13
	addq $30,72,$10
	addq $30,64,$11
	jsr $26,__builtin_new
	ldgp $29,0($26)
	bis $13,$13,$9
	ldt $f1,0($13)
	bis $13,$13,$12
	lda $1,$L489
	bis $13,$13,$2
	stq $1,96($30)
	stt $f1,72($30)
	stq $31,80($30)
	stq $10,88($30)
	stq $30,104($30)
	br $31,$L490
	.align 3
$L489:
	
$LSJ35:
	ldgp $29,$LSJ35-$L489($27)
	br $31,$L488
	.align 3
$L490:
	stq $10,0($2)
	bis $31,1,$1
	stq $1,0($0)
	ldq $1,0($9)
	ldt $f1,0($1)
	stt $f1,0($9)
	stq $0,0($11)
	br $31,$L494
	.align 3
$L488:
	bis $0,$0,$16
	jsr $26,__builtin_delete
	ldgp $29,0($26)
	jsr $26,__sjthrow
	ldgp $29,0($26)
	.align 3
$L494:
	ldq $2,0($13)
	addq $30,64,$6
	lda $1,_$_6Vector
	addq $30,120,$3
	bis $31,$31,$11
	ldt $f1,8($2)
	stq $1,128($30)
	stq $6,136($30)
	stt $f1,120($30)
	stq $3,8($2)
	.align 3
$L496:
	lda $1,6348
	cmpule $11,$1,$1
	beq $1,$L497
	ldq $1,64($30)
	bis $12,$12,$9
	addq $30,72,$10
	addq $30,64,$15
	ldq $1,0($1)
	cmpeq $1,1,$1
	bne $1,$L500
	lda $16,25408
	jsr $26,__builtin_new
	ldgp $29,0($26)
	ldt $f1,0($12)
	bis $12,$12,$2
	lda $1,$L502
	stq $1,96($30)
	stt $f1,72($30)
	stq $31,80($30)
	stq $10,88($30)
	stq $30,104($30)
	br $31,$L503
	.align 3
$L502:
	
$LSJ109:
	ldgp $29,$LSJ109-$L502($27)
	br $31,$L501
	.align 3
$L503:
	stq $10,0($2)
	bis $31,1,$1
	stq $1,0($0)
	bis $31,$31,$2
	ldq $1,0($9)
	lda $4,6348
	addq $0,8,$3
	ldt $f1,0($1)
	stt $f1,0($9)
	.align 3
$L508:
	ldq $1,0($15)
	s4addq $2,$1,$1
	addq $2,1,$2
	lds $f1,8($1)
	cmpule $2,$4,$1
	sts $f1,0($3)
	addq $3,4,$3
	bne $1,$L508
	ldq $2,0($15)
	ldq $1,0($2)
	subq $1,1,$1
	stq $1,0($2)
	stq $0,0($15)
$L500:
	ldq $2,0($15)
	bis $31,1,$1
	s4addq $11,8,$3
	stq $1,0($2)
	ldq $1,0($15)
	addq $1,$3,$1
	br $31,$L511
	.align 3
$L501:
	bis $0,$0,$16
	jsr $26,__builtin_delete
	ldgp $29,0($26)
	jsr $26,__sjthrow
	ldgp $29,0($26)
	.align 3
$L511:
	stl $11,0($1)
	addq $11,1,$11
	br $31,$L496
	.align 3
$L497:
	ldq $2,64($30)
	addq $30,72,$4
	lda $3,_$_6Vector
	addq $30,144,$5
	bis $31,$31,$10
	stq $2,72($30)
	ldq $1,0($2)
	addq $1,1,$1
	stq $1,0($2)
	ldq $1,0($13)
	ldt $f1,8($1)
	stq $3,152($30)
	stq $4,160($30)
	stt $f1,144($30)
	stq $5,8($1)
	.align 3
$L515:
	lda $1,6348
	cmpule $10,$1,$1
	beq $1,$L516
	ldq $1,72($30)
	bis $12,$12,$15
	addq $30,168,$9
	ldq $1,0($1)
	cmpeq $1,1,$1
	bne $1,$L519
	lda $16,25408
	jsr $26,__builtin_new
	ldgp $29,0($26)
	ldt $f1,0($12)
	bis $12,$12,$2
	lda $1,$L521
	stq $1,192($30)
	stt $f1,168($30)
	stq $31,176($30)
	stq $9,184($30)
	stq $30,200($30)
	br $31,$L522
	.align 3
$L521:
	
$LSJ240:
	ldgp $29,$LSJ240-$L521($27)
	br $31,$L520
	.align 3
$L522:
	stq $9,0($2)
	bis $31,1,$1
	stq $1,0($0)
	bis $31,$31,$2
	ldq $1,0($15)
	lda $4,6348
	addq $0,8,$3
	ldt $f1,0($1)
	stt $f1,0($15)
	.align 3
$L527:
	ldq $1,72($30)
	s4addq $2,$1,$1
	addq $2,1,$2
	lds $f1,8($1)
	cmpule $2,$4,$1
	sts $f1,0($3)
	addq $3,4,$3
	bne $1,$L527
	ldq $2,72($30)
	ldq $1,0($2)
	subq $1,1,$1
	stq $1,0($2)
	stq $0,72($30)
$L519:
	ldq $2,72($30)
	bis $31,1,$1
	s4addq $10,8,$3
	stq $1,0($2)
	ldq $1,72($30)
	addq $1,$3,$2
	br $31,$L530
	.align 3
$L520:
	bis $0,$0,$16
	jsr $26,__builtin_delete
	ldgp $29,0($26)
	jsr $26,__sjthrow
	ldgp $29,0($26)
	.align 3
$L530:
	bis $31,5,$1
	stl $1,0($2)
	addq $10,1,$10
	br $31,$L515
	.align 3
$L516:
	bis $31,$31,$11
	.align 3
$L532:
	lda $1,6348
	cmpule $11,$1,$1
	beq $1,$L533
	ldq $1,64($30)
	bis $12,$12,$9
	lda $14,cout
	addq $30,168,$10
	ldq $1,0($1)
	addq $30,64,$15
	cmpeq $1,1,$1
	bne $1,$L536
	lda $16,25408
	jsr $26,__builtin_new
	ldgp $29,0($26)
	ldt $f1,0($12)
	bis $12,$12,$2
	lda $1,$L538
	stq $1,192($30)
	stt $f1,168($30)
	stq $31,176($30)
	stq $10,184($30)
	stq $30,200($30)
	br $31,$L539
	.align 3
$L538:
	
$LSJ355:
	ldgp $29,$LSJ355-$L538($27)
	br $31,$L537
	.align 3
$L539:
	stq $10,0($2)
	bis $31,1,$1
	stq $1,0($0)
	bis $31,$31,$2
	ldq $1,0($9)
	lda $4,6348
	addq $0,8,$3
	ldt $f1,0($1)
	stt $f1,0($9)
	.align 3
$L544:
	ldq $1,0($15)
	s4addq $2,$1,$1
	addq $2,1,$2
	lds $f1,8($1)
	cmpule $2,$4,$1
	sts $f1,0($3)
	addq $3,4,$3
	bne $1,$L544
	ldq $2,0($15)
	ldq $1,0($2)
	subq $1,1,$1
	stq $1,0($2)
	stq $0,0($15)
$L536:
	ldq $2,0($15)
	bis $31,1,$1
	s4addq $11,8,$3
	stq $1,0($2)
	ldq $1,0($15)
	addq $1,$3,$1
	br $31,$L547
	.align 3
$L537:
	bis $0,$0,$16
	jsr $26,__builtin_delete
	ldgp $29,0($26)
	jsr $26,__sjthrow
	ldgp $29,0($26)
	.align 3
$L547:
	ldl $17,0($1)
	bis $14,$14,$16
	addq $11,1,$11
	jsr $26,__ls__7ostreami
	ldgp $29,0($26)
	lda $17,$LC0
	bis $0,$0,$16
	jsr $26,__ls__7ostreamPCc
	ldgp $29,0($26)
	br $31,$L532
	.align 3
$L533:
	lda $16,cout
	bis $31,$31,$10
	lda $27,endl__FR7ostream
	jsr $26,($27),0
	ldgp $29,0($26)
	.align 3
$L550:
	lda $1,6348
	cmpule $10,$1,$1
	beq $1,$L551
	ldq $1,72($30)
	bis $12,$12,$15
	lda $11,cout
	addq $30,168,$9
	ldq $1,0($1)
	cmpeq $1,1,$1
	bne $1,$L554
	lda $16,25408
	jsr $26,__builtin_new
	ldgp $29,0($26)
	ldt $f1,0($15)
	bis $15,$15,$2
	lda $1,$L556
	stq $1,192($30)
	stt $f1,168($30)
	stq $31,176($30)
	stq $9,184($30)
	stq $30,200($30)
	br $31,$L557
	.align 3
$L556:
	
$LSJ484:
	ldgp $29,$LSJ484-$L556($27)
	br $31,$L555
	.align 3
$L557:
	stq $9,0($2)
	bis $31,1,$1
	stq $1,0($0)
	bis $31,$31,$2
	ldq $1,0($15)
	lda $4,6348
	addq $0,8,$3
	ldt $f1,0($1)
	stt $f1,0($15)
	.align 3
$L562:
	ldq $1,72($30)
	s4addq $2,$1,$1
	addq $2,1,$2
	lds $f1,8($1)
	cmpule $2,$4,$1
	sts $f1,0($3)
	addq $3,4,$3
	bne $1,$L562
	ldq $2,72($30)
	ldq $1,0($2)
	subq $1,1,$1
	stq $1,0($2)
	stq $0,72($30)
$L554:
	ldq $2,72($30)
	bis $31,1,$1
	s4addq $10,8,$3
	stq $1,0($2)
	ldq $1,72($30)
	addq $1,$3,$1
	br $31,$L565
	.align 3
$L555:
	bis $0,$0,$16
	jsr $26,__builtin_delete
	ldgp $29,0($26)
	jsr $26,__sjthrow
	ldgp $29,0($26)
	.align 3
$L565:
	ldl $17,0($1)
	bis $11,$11,$16
	addq $10,1,$10
	jsr $26,__ls__7ostreami
	ldgp $29,0($26)
	lda $17,$LC0
	bis $0,$0,$16
	jsr $26,__ls__7ostreamPCc
	ldgp $29,0($26)
	br $31,$L550
	.align 3
$L551:
	lda $16,cout
	lda $27,endl__FR7ostream
	jsr $26,($27),0
	ldgp $29,0($26)
	ldq $2,0($13)
	ldq $1,8($2)
	ldt $f1,0($1)
	stt $f1,8($2)
	ldq $2,72($30)
	ldq $1,0($2)
	subq $1,1,$1
	stq $1,0($2)
	bne $1,$L572
	ldq $16,72($30)
	jsr $26,__builtin_delete
	ldgp $29,0($26)
$L572:
	ldq $2,0($13)
	ldq $1,8($2)
	ldt $f1,0($1)
	stt $f1,8($2)
	ldq $2,64($30)
	ldq $1,0($2)
	subq $1,1,$1
	stq $1,0($2)
	bne $1,$L577
	ldq $16,64($30)
	jsr $26,__builtin_delete
	ldgp $29,0($26)
$L577:
	bis $31,$31,$0
	ldq $26,0($30)
	ldq $9,8($30)
	ldq $10,16($30)
	ldq $11,24($30)
	ldq $12,32($30)
	ldq $13,40($30)
	ldq $14,48($30)
	ldq $15,56($30)
	addq $30,224,$30
	ret $31,($26),1
	.end main
	.align 3
	.ent _$_6Vector
_$_6Vector:
	ldgp $29,0($27)
$_$_6Vector..ng:
	lda $30,-32($30)
	.frame $30,32,$26,0
	stq $26,0($30)
	stq $9,8($30)
	stq $10,16($30)
	.mask 0x4000600,-32
	.prologue 1
	bis $16,$16,$9
	bis $17,$17,$10
	ldq $2,0($9)
	ldq $1,0($2)
	subq $1,1,$1
	stq $1,0($2)
	bne $1,$L482
	ldq $16,0($9)
	jsr $26,__builtin_delete
	ldgp $29,0($26)
$L482:
	bis $9,$9,$16
	and $10,1,$1
	beq $1,$L484
	jsr $26,__builtin_delete
	ldgp $29,0($26)
$L484:
	ldq $26,0($30)
	ldq $9,8($30)
	ldq $10,16($30)
	addq $30,32,$30
	ret $31,($26),1
	.end _$_6Vector

unless I'm mistaken, the loop test is
	lda $1, 6348
	cmpule $10,$1, $1
	beq $1, $L516

and the refcnt test is
	ldq $1,0($1)
	cmpeq $1,1,$1
	bne $1,$L519
	
- the refcnt test occurs inside the loop. It has not been moved outside
the loop by the compiler.
- the exceptional case (refcnt != 1) is inside the loop, and the normal
case (refcnt == 1) is coded as a forward conditional branch, predicted
to fall through (alpha architecture handbook).  

Did I miss something, or is the optimization truly as bad as it looks ?
Any optimization options I missed ? Any way to code the loop so that it
will run faster ?
-- 
	Marc Espie



More information about the Gcc mailing list