register __vector float vx3_r = {x[6], -x[6],x[6], -x[6]};\r
register __vector float vx3_i = {x[7], x[7],x[7], x[7]};\r
#endif\r
- register __vector float *vy = (__vector float *) y;\r
+ register __vector float *vptr_y = (__vector float *) y;\r
register __vector float *vptr_a0 = (__vector float *) a0;\r
register __vector float *vptr_a1 = (__vector float *) a1;\r
register __vector float *vptr_a2 = (__vector float *) a2;\r
register __vector float *vptr_a3 = (__vector float *) a3; \r
BLASLONG i = 0; \r
- for (;i< n / 2; i+=2) {\r
- register __vector float vy_0 = vy[i];\r
- register __vector float vy_1 = vy[i + 1];\r
- register __vector float va0 = vptr_a0[i];\r
- register __vector float va1 = vptr_a1[i];\r
- register __vector float va2 = vptr_a2[i];\r
- register __vector float va3 = vptr_a3[i];\r
- register __vector float va0_1 = vptr_a0[i + 1];\r
- register __vector float va1_1 = vptr_a1[i + 1];\r
- register __vector float va2_1 = vptr_a2[i + 1];\r
- register __vector float va3_1 = vptr_a3[i + 1];\r
+ BLASLONG i2=16;\r
+ for (;i< n * 8; i+=32,i2+=32) {\r
+ register __vector float vy_0 = vec_vsx_ld(i,vptr_y);\r
+ register __vector float vy_1 = vec_vsx_ld(i2,vptr_y);\r
+ register __vector float va0 = vec_vsx_ld(i,vptr_a0);\r
+ register __vector float va1 = vec_vsx_ld(i, vptr_a1);\r
+ register __vector float va2 = vec_vsx_ld(i ,vptr_a2);\r
+ register __vector float va3 = vec_vsx_ld(i ,vptr_a3);\r
+ register __vector float va0_1 = vec_vsx_ld(i2 ,vptr_a0);\r
+ register __vector float va1_1 = vec_vsx_ld(i2 ,vptr_a1);\r
+ register __vector float va2_1 = vec_vsx_ld(i2 ,vptr_a2);\r
+ register __vector float va3_1 = vec_vsx_ld(i2 ,vptr_a3);\r
\r
vy_0 += va0*vx0_r + va1*vx1_r + va2*vx2_r + va3*vx3_r;\r
vy_1 += va0_1*vx0_r + va1_1*vx1_r + va2_1*vx2_r + va3_1*vx3_r;\r
vy_0 += va0*vx0_i + va1*vx1_i + va2*vx2_i + va3*vx3_i;\r
vy_1 += va0_1*vx0_i + va1_1*vx1_i + va2_1*vx2_i + va3_1*vx3_i;\r
\r
- vy[i] = vy_0;\r
- vy[i + 1] = vy_1;\r
+ vec_vsx_st(vy_0 ,i, vptr_y);\r
+ vec_vsx_st(vy_1,i2,vptr_y);\r
}\r
\r
} \r
register __vector float vx1_r = {x[2], -x[2],x[2], -x[2]};\r
register __vector float vx1_i = {x[3], x[3],x[3], x[3]}; \r
#endif\r
- register __vector float *vy = (__vector float *) y;\r
+ register __vector float *vptr_y = (__vector float *) y;\r
register __vector float *vptr_a0 = (__vector float *) a0;\r
register __vector float *vptr_a1 = (__vector float *) a1; \r
- BLASLONG i = 0; \r
- for (;i< n / 2; i+=2) {\r
- register __vector float vy_0 = vy[i];\r
- register __vector float vy_1 = vy[i + 1];\r
- register __vector float va0 = vptr_a0[i];\r
- register __vector float va1 = vptr_a1[i]; \r
- register __vector float va0_1 = vptr_a0[i + 1];\r
- register __vector float va1_1 = vptr_a1[i + 1]; \r
+ BLASLONG i = 0;\r
+ BLASLONG i2 = 16; \r
+ for (;i< n * 8; i+=32, i2+=32) { \r
+ register __vector float vy_0 = vec_vsx_ld(i,vptr_y);\r
+ register __vector float vy_1 = vec_vsx_ld(i2,vptr_y);\r
+ register __vector float va0 = vec_vsx_ld(i,vptr_a0);\r
+ register __vector float va1 = vec_vsx_ld(i, vptr_a1); \r
+ register __vector float va0_1 = vec_vsx_ld(i2 ,vptr_a0);\r
+ register __vector float va1_1 = vec_vsx_ld(i2 ,vptr_a1); \r
+\r
register __vector float va0x = vec_perm(va0, va0,swap_mask);\r
register __vector float va0x_1 = vec_perm(va0_1, va0_1,swap_mask);\r
register __vector float va1x = vec_perm(va1, va1,swap_mask);\r
vy_0 += va0*vx0_r + va1*vx1_r + va0x*vx0_i + va1x*vx1_i;\r
vy_1 += va0_1*vx0_r + va1_1*vx1_r + va0x_1*vx0_i + va1x_1*vx1_i; \r
\r
- vy[i] = vy_0;\r
- vy[i + 1] = vy_1;\r
+ vec_vsx_st(vy_0 ,i, vptr_y);\r
+ vec_vsx_st(vy_1,i2,vptr_y);\r
}\r
\r
}\r
register __vector float vx0_r = {x[0], -x[0],x[0], -x[0]};\r
register __vector float vx0_i = {x[1], x[1],x[1], x[1]}; \r
#endif\r
- register __vector float *vy = (__vector float *) y;\r
+ register __vector float *vptr_y = (__vector float *) y;\r
register __vector float *vptr_a0 = (__vector float *) ap; \r
BLASLONG i = 0; \r
- for (;i< n / 2; i+=2) {\r
- register __vector float vy_0 = vy[i];\r
- register __vector float vy_1 = vy[i + 1];\r
- register __vector float va0 = vptr_a0[i];\r
- register __vector float va0_1 = vptr_a0[i + 1]; \r
+ BLASLONG i2 = 16; \r
+ for (;i< n * 8; i+=32, i2+=32) { \r
+ register __vector float vy_0 = vec_vsx_ld(i,vptr_y);\r
+ register __vector float vy_1 = vec_vsx_ld(i2,vptr_y);\r
+ register __vector float va0 = vec_vsx_ld(i,vptr_a0); \r
+ register __vector float va0_1 = vec_vsx_ld(i2 ,vptr_a0); \r
+\r
register __vector float va0x = vec_perm(va0, va0,swap_mask);\r
register __vector float va0x_1 = vec_perm(va0_1, va0_1,swap_mask);\r
vy_0 += va0*vx0_r + va0x*vx0_i;\r
vy_1 += va0_1*vx0_r + va0x_1*vx0_i; \r
\r
- vy[i] = vy_0;\r
- vy[i + 1] = vy_1;\r
+ vec_vsx_st(vy_0 ,i, vptr_y);\r
+ vec_vsx_st(vy_1,i2,vptr_y);\r
}\r
}\r
\r
\r
register __vector float *vptr_src = (__vector float *) src;\r
register __vector float *vptr_y = (__vector float *) dest; \r
- for (i = 0; i < n/2; i += 2 ){\r
\r
- register __vector float vy_0 = vptr_y[i];\r
- register __vector float vy_1 = vptr_y[i +1]; \r
+ BLASLONG i2 = 16; \r
+ for (;i< n * 8; i+=32, i2+=32) { \r
+ register __vector float vy_0 = vec_vsx_ld(i,vptr_y);\r
+ register __vector float vy_1 = vec_vsx_ld(i2,vptr_y);\r
+\r
+\r
+ register __vector float vsrc = vec_vsx_ld(i,vptr_src);\r
+ register __vector float vsrc_1 = vec_vsx_ld(i2,vptr_src);\r
+\r
+ register __vector float vsrcx = vec_perm(vsrc, vsrc, swap_mask);\r
+ register __vector float vsrcx_1 = vec_perm(vsrc_1, vsrc_1, swap_mask);\r
\r
- register __vector float vsrc = vptr_src[i];\r
- register __vector float vsrc_1 = vptr_src[i + 1]; \r
- register __vector float vsrcx = vec_perm(vsrc, vsrc, swap_mask);\r
- register __vector float vsrcx_1 = vec_perm(vsrc_1, vsrc_1, swap_mask);\r
+ vy_0 += vsrc*valpha_r + vsrcx*valpha_i;\r
+ vy_1 += vsrc_1*valpha_r + vsrcx_1*valpha_i; \r
\r
- vy_0 += vsrc*valpha_r + vsrcx*valpha_i;\r
- vy_1 += vsrc_1*valpha_r + vsrcx_1*valpha_i; \r
- vptr_y[i] = vy_0;\r
- vptr_y[i+1 ] = vy_1; \r
+ vec_vsx_st(vy_0 ,i, vptr_y);\r
+ vec_vsx_st(vy_1,i2,vptr_y);\r
\r
}\r
\r
static const unsigned char __attribute__((aligned(16))) swap_mask_arr[]={ 4,5,6,7,0,1,2,3, 12,13,14,15, 8,9,10,11};\r
\r
static void cgemv_kernel_4x4(BLASLONG n, BLASLONG lda, FLOAT *ap, FLOAT *x, FLOAT *y, FLOAT alpha_r, FLOAT alpha_i) {\r
- BLASLONG i;\r
+\r
FLOAT *a0, *a1, *a2, *a3;\r
a0 = ap;\r
a1 = ap + lda;\r
register __vector float vtemp2_r = {0.0, 0.0,0.0,0.0};\r
register __vector float vtemp3_p = {0.0, 0.0,0.0,0.0};\r
register __vector float vtemp3_r = {0.0, 0.0,0.0,0.0};\r
- __vector float* va0 = (__vector float*) a0;\r
- __vector float* va1 = (__vector float*) a1;\r
- __vector float* va2 = (__vector float*) a2;\r
- __vector float* va3 = (__vector float*) a3;\r
+ __vector float* vptr_a0 = (__vector float*) a0;\r
+ __vector float* vptr_a1 = (__vector float*) a1;\r
+ __vector float* vptr_a2 = (__vector float*) a2;\r
+ __vector float* vptr_a3 = (__vector float*) a3;\r
__vector float* v_x = (__vector float*) x;\r
\r
- for (i = 0; i < n / 2; i+=2) {\r
- register __vector float vx_0 = v_x[i]; \r
- register __vector float vx_1 = v_x[i+1]; \r
+ BLASLONG i = 0;\r
+ BLASLONG i2 = 16; \r
+ for (;i< n * 8; i+=32, i2+=32) { \r
+ register __vector float vx_0 = vec_vsx_ld( i,v_x) ; \r
+ register __vector float vx_1 = vec_vsx_ld(i2, v_x); \r
+ \r
register __vector float vxr_0 = vec_perm(vx_0, vx_0, swap_mask);\r
register __vector float vxr_1 = vec_perm(vx_1, vx_1, swap_mask);\r
\r
- vtemp0_p += vx_0*va0[i] + vx_1*va0[i+1] ;\r
- vtemp0_r += vxr_0*va0[i] + vxr_1*va0[i+1]; \r
- vtemp1_p += vx_0*va1[i] + vx_1*va1[i+1];\r
- vtemp1_r += vxr_0*va1[i] + vxr_1*va1[i+1]; \r
- vtemp2_p += vx_0*va2[i] + vx_1*va2[i+1];\r
- vtemp2_r += vxr_0*va2[i] + vxr_1*va2[i+1]; \r
- vtemp3_p += vx_0*va3[i] + vx_1*va3[i+1];\r
- vtemp3_r += vxr_0*va3[i] + vxr_1*va3[i+1]; \r
+ register __vector float va0 = vec_vsx_ld(i,vptr_a0);\r
+ register __vector float va1 = vec_vsx_ld(i, vptr_a1);\r
+ register __vector float va2 = vec_vsx_ld(i ,vptr_a2);\r
+ register __vector float va3 = vec_vsx_ld(i ,vptr_a3);\r
+ register __vector float va0_1 = vec_vsx_ld(i2 ,vptr_a0);\r
+ register __vector float va1_1 = vec_vsx_ld(i2 ,vptr_a1);\r
+ register __vector float va2_1 = vec_vsx_ld(i2 ,vptr_a2);\r
+ register __vector float va3_1 = vec_vsx_ld(i2 ,vptr_a3);\r
+\r
+\r
+ vtemp0_p += vx_0*va0 + vx_1*va0_1 ;\r
+ vtemp0_r += vxr_0*va0 + vxr_1*va0_1; \r
+ vtemp1_p += vx_0*va1 + vx_1*va1_1;\r
+ vtemp1_r += vxr_0*va1 + vxr_1*va1_1; \r
+ vtemp2_p += vx_0*va2 + vx_1*va2_1;\r
+ vtemp2_r += vxr_0*va2 + vxr_1*va2_1; \r
+ vtemp3_p += vx_0*va3 + vx_1*va3_1;\r
+ vtemp3_r += vxr_0*va3 + vxr_1*va3_1; \r
\r
}\r
\r
\r
\r
static void cgemv_kernel_4x2(BLASLONG n, BLASLONG lda, FLOAT *ap, FLOAT *x, FLOAT *y, FLOAT alpha_r, FLOAT alpha_i) {\r
- BLASLONG i;\r
+\r
FLOAT *a0, *a1;\r
a0 = ap;\r
a1 = ap + lda; \r
register __vector float vtemp0_r = {0.0, 0.0,0.0,0.0};\r
register __vector float vtemp1_p = {0.0, 0.0,0.0,0.0};\r
register __vector float vtemp1_r = {0.0, 0.0,0.0,0.0}; \r
- __vector float* va0 = (__vector float*) a0;\r
- __vector float* va1 = (__vector float*) a1; \r
+\r
+\r
+ __vector float* vptr_a0 = (__vector float*) a0;\r
+ __vector float* vptr_a1 = (__vector float*) a1; \r
__vector float* v_x = (__vector float*) x;\r
\r
- for (i = 0; i < n / 2; i+=2) {\r
- register __vector float vx_0 = v_x[i]; \r
- register __vector float vx_1 = v_x[i+1]; \r
+ BLASLONG i = 0;\r
+ BLASLONG i2 = 16; \r
+ for (;i< n * 8; i+=32, i2+=32) { \r
+ register __vector float vx_0 = vec_vsx_ld( i,v_x) ; \r
+ register __vector float vx_1 = vec_vsx_ld(i2, v_x); \r
+ \r
register __vector float vxr_0 = vec_perm(vx_0, vx_0, swap_mask);\r
register __vector float vxr_1 = vec_perm(vx_1, vx_1, swap_mask);\r
\r
- vtemp0_p += vx_0*va0[i] + vx_1*va0[i+1] ;\r
- vtemp0_r += vxr_0*va0[i] + vxr_1*va0[i+1]; \r
- vtemp1_p += vx_0*va1[i] + vx_1*va1[i+1];\r
- vtemp1_r += vxr_0*va1[i] + vxr_1*va1[i+1]; \r
+ register __vector float va0 = vec_vsx_ld(i,vptr_a0);\r
+ register __vector float va1 = vec_vsx_ld(i, vptr_a1); \r
+ register __vector float va0_1 = vec_vsx_ld(i2 ,vptr_a0);\r
+ register __vector float va1_1 = vec_vsx_ld(i2 ,vptr_a1); \r
\r
- }\r
\r
+ vtemp0_p += vx_0*va0 + vx_1*va0_1 ;\r
+ vtemp0_r += vxr_0*va0 + vxr_1*va0_1; \r
+ vtemp1_p += vx_0*va1 + vx_1*va1_1;\r
+ vtemp1_r += vxr_0*va1 + vxr_1*va1_1; \r
+\r
+ }\r
#if ( !defined(CONJ) && !defined(XCONJ) ) || ( defined(CONJ) && defined(XCONJ) )\r
\r
register FLOAT temp_r0 = vtemp0_p[0] - vtemp0_p[1] + vtemp0_p[2] - vtemp0_p[3];\r
\r
\r
static void cgemv_kernel_4x1(BLASLONG n, FLOAT *ap, FLOAT *x, FLOAT *y, FLOAT alpha_r, FLOAT alpha_i) {\r
- BLASLONG i; \r
+ \r
__vector unsigned char swap_mask = *((__vector unsigned char*)swap_mask_arr);\r
//p for positive(real*real,image*image,real*real,image*image) r for image (real*image,image*real,real*image,image*real)\r
register __vector float vtemp0_p = {0.0, 0.0,0.0,0.0};\r
register __vector float vtemp0_r = {0.0, 0.0,0.0,0.0}; \r
- __vector float* va0 = (__vector float*) ap; \r
+ __vector float* vptr_a0 = (__vector float*) ap; \r
__vector float* v_x = (__vector float*) x;\r
-\r
- for (i = 0; i < n / 2; i+=2) {\r
- register __vector float vx_0 = v_x[i]; \r
- register __vector float vx_1 = v_x[i+1]; \r
+ BLASLONG i = 0;\r
+ BLASLONG i2 = 16; \r
+ for (;i< n * 8; i+=32, i2+=32) { \r
+ register __vector float vx_0 = vec_vsx_ld( i,v_x) ; \r
+ register __vector float vx_1 = vec_vsx_ld(i2, v_x); \r
+ \r
register __vector float vxr_0 = vec_perm(vx_0, vx_0, swap_mask);\r
register __vector float vxr_1 = vec_perm(vx_1, vx_1, swap_mask);\r
\r
- vtemp0_p += vx_0*va0[i] + vx_1*va0[i+1] ;\r
- vtemp0_r += vxr_0*va0[i] + vxr_1*va0[i+1]; \r
+ register __vector float va0 = vec_vsx_ld(i,vptr_a0); \r
+ register __vector float va0_1 = vec_vsx_ld(i2 ,vptr_a0); \r
\r
+ vtemp0_p += vx_0*va0 + vx_1*va0_1 ;\r
+ vtemp0_r += vxr_0*va0 + vxr_1*va0_1; \r
}\r
\r
#if ( !defined(CONJ) && !defined(XCONJ) ) || ( defined(CONJ) && defined(XCONJ) )\r