|
123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650651652653654655656657658659660661662663664665666667668669670671672673674675676677678679680681682683684685686687688689690691692693694695696697698699700701702703704705706707708709710711712713714715716717718719720721722723724725726727728729730731732733734735736737738739740741742743744745746747748749750751752753754755756757758759760761762763764765766767768769770771772773774775776777778779780781782783784785786787788789790791792793794795796797798799800801802803804805806807808809810811812813814815816817818819820821822823824825826827828829830831832833834835836837838839840841842843844845846847848849850851852853854855856857858859860861862863864865866867868869870871872873874875876877878879880881882883884885 |
- #include "common.h"
-
- #define MADD_ALPHA_N_STORE(C, res, alpha) \
- C[0] = res ## _r * alpha ## _r - res ## _i * alpha ## _i; \
- C[1] = res ## _r * alpha ## _i + res ## _i * alpha ## _r;
-
- #if defined(NN) || defined(NT) || defined(TN) || defined(TT)
- #define MADD(res, op1, op2) \
- res ## _r += op1 ## _r * op2 ## _r; \
- res ## _r -= op1 ## _i * op2 ## _i; \
- res ## _i += op1 ## _r * op2 ## _i; \
- res ## _i += op1 ## _i * op2 ## _r;
- #elif defined(NR) || defined(NC) || defined(TR) || defined(TC)
- #define MADD(res, op1, op2) \
- res ## _r += op1 ## _r * op2 ## _r; \
- res ## _r += op1 ## _i * op2 ## _i; \
- res ## _i -= op1 ## _r * op2 ## _i; \
- res ## _i += op1 ## _i * op2 ## _r;
- #elif defined(RN) || defined(RT) || defined(CN) || defined(CT)
- #define MADD(res, op1, op2) \
- res ## _r += op1 ## _r * op2 ## _r; \
- res ## _r += op1 ## _i * op2 ## _i; \
- res ## _i += op1 ## _r * op2 ## _i; \
- res ## _i -= op1 ## _i * op2 ## _r;
- #elif defined(RR) || defined(RC) || defined(CR) || defined(CC)
- #define MADD(res, op1, op2) \
- res ## _r += op1 ## _r * op2 ## _r; \
- res ## _r -= op1 ## _i * op2 ## _i; \
- res ## _i -= op1 ## _r * op2 ## _i; \
- res ## _i -= op1 ## _i * op2 ## _r;
- #endif
-
- int CNAME(BLASLONG bm,BLASLONG bn,BLASLONG bk,FLOAT alpha_r, FLOAT alpha_i,FLOAT* ba,FLOAT* bb,FLOAT* C,BLASLONG ldc
- , BLASLONG offset
- )
- {
-
- BLASLONG i,j,k;
- FLOAT *C0,*C1,*C2,*C3,*ptrba,*ptrbb;
- FLOAT res00_r, res01_r, res02_r, res03_r;
- FLOAT res00_i, res01_i, res02_i, res03_i;
- FLOAT res10_r, res11_r, res12_r, res13_r;
- FLOAT res10_i, res11_i, res12_i, res13_i;
- FLOAT res20_r, res21_r, res22_r, res23_r;
- FLOAT res20_i, res21_i, res22_i, res23_i;
- FLOAT res30_r, res31_r, res32_r, res33_r;
- FLOAT res30_i, res31_i, res32_i, res33_i;
- FLOAT a0_r, a1_r;
- FLOAT a0_i, a1_i;
- FLOAT b0_r, b1_r, b2_r, b3_r;
- FLOAT b0_i, b1_i, b2_i, b3_i;
- BLASLONG off, temp;
-
- #if defined(TRMMKERNEL) && !defined(LEFT)
- off = -offset;
- #else
- off = 0;
- #endif
-
- for (j=0; j<bn/4; j+=1) // do blocks of the Mx4 loops
- {
- C0 = C;
- C1 = C0+2*ldc;
- C2 = C1+2*ldc;
- C3 = C2+2*ldc;
-
-
- #if defined(TRMMKERNEL) && defined(LEFT)
- off = offset;
- #endif
-
- ptrba = ba;
-
- for (i=0; i<bm/4; i+=1) // do blocks of 4x4
- {
-
- #if (defined(LEFT) && defined(TRANSA)) || (!defined(LEFT) && !defined(TRANSA))
- ptrbb = bb;
- #else
-
- ptrba += off*4*2; // number of values in A
- ptrbb = bb + off*4*2; // number of values in B
- #endif
-
- res00_r = 0;
- res00_i = 0;
- res01_r = 0;
- res01_i = 0;
- res02_r = 0;
- res02_i = 0;
- res03_r = 0;
- res03_i = 0;
-
- res10_r = 0;
- res10_i = 0;
- res11_r = 0;
- res11_i = 0;
- res12_r = 0;
- res12_i = 0;
- res13_r = 0;
- res13_i = 0;
-
- res20_r = 0;
- res20_i = 0;
- res21_r = 0;
- res21_i = 0;
- res22_r = 0;
- res22_i = 0;
- res23_r = 0;
- res23_i = 0;
-
- res30_r = 0;
- res30_i = 0;
- res31_r = 0;
- res31_i = 0;
- res32_r = 0;
- res32_i = 0;
- res33_r = 0;
- res33_i = 0;
-
- #if (defined(LEFT) && !defined(TRANSA)) || (!defined(LEFT) && defined(TRANSA))
- temp = bk - off;
- #elif defined(LEFT)
- temp = off + 4;
- #else
- temp = off + 4;
- #endif
-
- for (k=0; k<temp; k++)
- {
- b0_r = ptrbb[2*0+0]; b0_i = ptrbb[2*0+1];
- b1_r = ptrbb[2*1+0]; b1_i = ptrbb[2*1+1];
- b2_r = ptrbb[2*2+0]; b2_i = ptrbb[2*2+1];
- b3_r = ptrbb[2*3+0]; b3_i = ptrbb[2*3+1];
-
- a0_r = ptrba[2*0+0]; a0_i = ptrba[2*0+1];
- MADD(res00, a0, b0);
- MADD(res10, a0, b1);
- MADD(res20, a0, b2);
- MADD(res30, a0, b3);
-
- a1_r = ptrba[2*1+0]; a1_i = ptrba[2*1+1];
- MADD(res01, a1, b0);
- MADD(res11, a1, b1);
- MADD(res21, a1, b2);
- MADD(res31, a1, b3);
-
- a0_r = ptrba[2*2+0]; a0_i = ptrba[2*2+1];
- MADD(res02, a0, b0);
- MADD(res12, a0, b1);
- MADD(res22, a0, b2);
- MADD(res32, a0, b3);
-
-
- a1_r = ptrba[2*3+0]; a1_i = ptrba[2*3+1];
- MADD(res03, a1, b0);
- MADD(res13, a1, b1);
- MADD(res23, a1, b2);
- MADD(res33, a1, b3);
-
- ptrba = ptrba+8;
- ptrbb = ptrbb+8;
- }
-
- MADD_ALPHA_N_STORE(C0, res00, alpha);
- C0 = C0 + 2;
- MADD_ALPHA_N_STORE(C0, res01, alpha);
- C0 = C0 + 2;
- MADD_ALPHA_N_STORE(C0, res02, alpha);
- C0 = C0 + 2;
- MADD_ALPHA_N_STORE(C0, res03, alpha);
- C0 = C0 + 2;
-
- MADD_ALPHA_N_STORE(C1, res10, alpha);
- C1 = C1 + 2;
- MADD_ALPHA_N_STORE(C1, res11, alpha);
- C1 = C1 + 2;
- MADD_ALPHA_N_STORE(C1, res12, alpha);
- C1 = C1 + 2;
- MADD_ALPHA_N_STORE(C1, res13, alpha);
- C1 = C1 + 2;
-
- MADD_ALPHA_N_STORE(C2, res20, alpha);
- C2 = C2 + 2;
- MADD_ALPHA_N_STORE(C2, res21, alpha);
- C2 = C2 + 2;
- MADD_ALPHA_N_STORE(C2, res22, alpha);
- C2 = C2 + 2;
- MADD_ALPHA_N_STORE(C2, res23, alpha);
- C2 = C2 + 2;
-
- MADD_ALPHA_N_STORE(C3, res30, alpha);
- C3 = C3 + 2;
- MADD_ALPHA_N_STORE(C3, res31, alpha);
- C3 = C3 + 2;
- MADD_ALPHA_N_STORE(C3, res32, alpha);
- C3 = C3 + 2;
- MADD_ALPHA_N_STORE(C3, res33, alpha);
- C3 = C3 + 2;
-
-
- #if ( defined(LEFT) && defined(TRANSA)) || (!defined(LEFT) && !defined(TRANSA))
- temp = bk-off;
- #if defined(LEFT)
- temp = temp - 4;
- #else
- temp = temp - 4;
- #endif
- ptrba += temp*4*2; // number of values in A
- ptrbb += temp*4*2; // number of values in B
- #endif
- #ifdef LEFT
- off += 4; // number of values in A
- #endif
-
-
- }
-
- if ( bm & 2 ) // do any 2x4 loop
- {
-
- #if (defined(LEFT) && defined(TRANSA)) || (!defined(LEFT) && !defined(TRANSA))
- ptrbb = bb;
- #else
- ptrba += off*2*2;
- ptrbb = bb + off*4*2;
- #endif
-
-
- res00_r = 0;
- res00_i = 0;
- res01_r = 0;
- res01_i = 0;
-
- res10_r = 0;
- res10_i = 0;
- res11_r = 0;
- res11_i = 0;
-
- res20_r = 0;
- res20_i = 0;
- res21_r = 0;
- res21_i = 0;
-
- res30_r = 0;
- res30_i = 0;
- res31_r = 0;
- res31_i = 0;
-
- #if (defined(LEFT) && !defined(TRANSA)) || (!defined(LEFT) && defined(TRANSA))
- temp = bk-off;
- #elif defined(LEFT)
- temp = off+2; // number of values in A
- #else
- temp = off+4; // number of values in B
- #endif
-
- for (k=0; k<temp; k++)
- {
- b0_r = ptrbb[2*0+0]; b0_i = ptrbb[2*0+1];
- b1_r = ptrbb[2*1+0]; b1_i = ptrbb[2*1+1];
- b2_r = ptrbb[2*2+0]; b2_i = ptrbb[2*2+1];
- b3_r = ptrbb[2*3+0]; b3_i = ptrbb[2*3+1];
-
- a0_r = ptrba[2*0+0]; a0_i = ptrba[2*0+1];
- MADD(res00, a0, b0);
- MADD(res10, a0, b1);
- MADD(res20, a0, b2);
- MADD(res30, a0, b3);
-
- a1_r = ptrba[2*1+0]; a1_i = ptrba[2*1+1];
- MADD(res01, a1, b0);
- MADD(res11, a1, b1);
- MADD(res21, a1, b2);
- MADD(res31, a1, b3);
-
-
- ptrba = ptrba+4;
- ptrbb = ptrbb+8;
- }
-
- MADD_ALPHA_N_STORE(C0, res00, alpha);
- C0 = C0 + 2;
- MADD_ALPHA_N_STORE(C0, res01, alpha);
- C0 = C0 + 2;
-
- MADD_ALPHA_N_STORE(C1, res10, alpha);
- C1 = C1 + 2;
- MADD_ALPHA_N_STORE(C1, res11, alpha);
- C1 = C1 + 2;
-
- MADD_ALPHA_N_STORE(C2, res20, alpha);
- C2 = C2 + 2;
- MADD_ALPHA_N_STORE(C2, res21, alpha);
- C2 = C2 + 2;
-
- MADD_ALPHA_N_STORE(C3, res30, alpha);
- C3 = C3 + 2;
- MADD_ALPHA_N_STORE(C3, res31, alpha);
- C3 = C3 + 2;
-
-
-
-
-
- #if ( defined(LEFT) && defined(TRANSA)) || (!defined(LEFT) && !defined(TRANSA))
- temp = bk - off;
- #ifdef LEFT
- temp -= 2; // number of values in A
- #else
- temp -= 4; // number of values in B
- #endif
- ptrba += temp*2*2;
- ptrbb += temp*4*2;
- #endif
-
- #ifdef LEFT
- off += 2; // number of values in A
- #endif
-
-
- }
-
- if ( bm & 1 ) // do any 1x4 loop
- {
-
- #if (defined(LEFT) && defined(TRANSA)) || (!defined(LEFT) && !defined(TRANSA))
- ptrbb = bb;
- #else
- ptrba += off*1*2;
- ptrbb = bb + off*4*2;
- #endif
-
- res00_r = 0;
- res00_i = 0;
- res10_r = 0;
- res10_i = 0;
- res20_r = 0;
- res20_i = 0;
- res30_r = 0;
- res30_i = 0;
-
-
- #if (defined(LEFT) && !defined(TRANSA)) || (!defined(LEFT) && defined(TRANSA))
- temp = bk-off;
- #elif defined(LEFT)
- temp = off+1; // number of values in A
- #else
- temp = off+4; // number of values in B
- #endif
-
- for (k=0; k<temp; k++)
- {
- b0_r = ptrbb[2*0+0]; b0_i = ptrbb[2*0+1];
- b1_r = ptrbb[2*1+0]; b1_i = ptrbb[2*1+1];
- b2_r = ptrbb[2*2+0]; b2_i = ptrbb[2*2+1];
- b3_r = ptrbb[2*3+0]; b3_i = ptrbb[2*3+1];
-
- a0_r = ptrba[2*0+0]; a0_i = ptrba[2*0+1];
- MADD(res00, a0, b0);
- MADD(res10, a0, b1);
- MADD(res20, a0, b2);
- MADD(res30, a0, b3);
-
-
- ptrba = ptrba+2;
- ptrbb = ptrbb+8;
- }
-
- MADD_ALPHA_N_STORE(C0, res00, alpha);
- C0 = C0 + 2;
-
- MADD_ALPHA_N_STORE(C1, res10, alpha);
- C1 = C1 + 2;
-
- MADD_ALPHA_N_STORE(C2, res20, alpha);
- C2 = C2 + 2;
-
- MADD_ALPHA_N_STORE(C3, res30, alpha);
- C3 = C3 + 2;
-
-
- #if ( defined(LEFT) && defined(TRANSA)) || (!defined(LEFT) && !defined(TRANSA))
- temp = bk - off;
- #ifdef LEFT
- temp -= 1; // number of values in A
- #else
- temp -= 4; // number of values in B
- #endif
- ptrba += temp*1*2;
- ptrbb += temp*4*2;
- #endif
-
- #ifdef LEFT
- off += 1; // number of values in A
- #endif
-
-
- }
-
-
- #if defined(TRMMKERNEL) && !defined(LEFT)
- off += 4;
- #endif
-
- k = (bk<<3);
- bb = bb+k;
- i = (ldc<<3);
- C = C+i;
- }
-
- for (j=0; j<(bn&2); j+=2) // do the Mx2 loops
- {
- C0 = C;
- C1 = C0+ldc*2;
-
- #if defined(TRMMKERNEL) && defined(LEFT)
- off = offset;
- #endif
-
-
- ptrba = ba;
-
- for (i=0; i<bm/4; i+=1) // do blocks of 4x2
- {
-
- #if (defined(LEFT) && defined(TRANSA)) || (!defined(LEFT) && !defined(TRANSA))
- ptrbb = bb;
- #else
- ptrba += off*4*2;
- ptrbb = bb + off*2*2;
- #endif
-
- res00_r = 0;
- res00_i = 0;
- res01_r = 0;
- res01_i = 0;
- res02_r = 0;
- res02_i = 0;
- res03_r = 0;
- res03_i = 0;
-
- res10_r = 0;
- res10_i = 0;
- res11_r = 0;
- res11_i = 0;
- res12_r = 0;
- res12_i = 0;
- res13_r = 0;
- res13_i = 0;
-
-
- #if (defined(LEFT) && !defined(TRANSA)) || (!defined(LEFT) && defined(TRANSA))
- temp = bk-off;
- #elif defined(LEFT)
- temp = off+4; // number of values in A
- #else
- temp = off+2; // number of values in B
- #endif
-
- for (k=0; k<temp; k++)
- {
- b0_r = ptrbb[2*0+0]; b0_i = ptrbb[2*0+1];
- b1_r = ptrbb[2*1+0]; b1_i = ptrbb[2*1+1];
-
- a0_r = ptrba[2*0+0]; a0_i = ptrba[2*0+1];
- MADD(res00, a0, b0);
- MADD(res10, a0, b1);
-
- a1_r = ptrba[2*1+0]; a1_i = ptrba[2*1+1];
- MADD(res01, a1, b0);
- MADD(res11, a1, b1);
-
- a0_r = ptrba[2*2+0]; a0_i = ptrba[2*2+1];
- MADD(res02, a0, b0);
- MADD(res12, a0, b1);
-
- a1_r = ptrba[2*3+0]; a1_i = ptrba[2*3+1];
- MADD(res03, a1, b0);
- MADD(res13, a1, b1);
-
- ptrba = ptrba+8;
- ptrbb = ptrbb+4;
- }
-
- MADD_ALPHA_N_STORE(C0, res00, alpha);
- C0 = C0 + 2;
- MADD_ALPHA_N_STORE(C0, res01, alpha);
- C0 = C0 + 2;
- MADD_ALPHA_N_STORE(C0, res02, alpha);
- C0 = C0 + 2;
- MADD_ALPHA_N_STORE(C0, res03, alpha);
- C0 = C0 + 2;
-
- MADD_ALPHA_N_STORE(C1, res10, alpha);
- C1 = C1 + 2;
- MADD_ALPHA_N_STORE(C1, res11, alpha);
- C1 = C1 + 2;
- MADD_ALPHA_N_STORE(C1, res12, alpha);
- C1 = C1 + 2;
- MADD_ALPHA_N_STORE(C1, res13, alpha);
- C1 = C1 + 2;
-
-
- #if ( defined(LEFT) && defined(TRANSA)) || (!defined(LEFT) && !defined(TRANSA))
- temp = bk - off;
- #ifdef LEFT
- temp -= 4; // number of values in A
- #else
- temp -= 2; // number of values in B
- #endif
- ptrba += temp*4*2;
- ptrbb += temp*2*2;
- #endif
-
- #ifdef LEFT
- off += 4; // number of values in A
- #endif
-
- }
-
- if ( bm & 2 ) // do any 2x2 loop
- {
-
- #if (defined(LEFT) && defined(TRANSA)) || (!defined(LEFT) && !defined(TRANSA))
- ptrbb = bb;
- #else
- ptrba += off*2*2;
- ptrbb = bb + off*2*2;
- #endif
-
- res00_r = 0;
- res00_i = 0;
- res01_r = 0;
- res01_i = 0;
-
- res10_r = 0;
- res10_i = 0;
- res11_r = 0;
- res11_i = 0;
-
-
- #if (defined(LEFT) && !defined(TRANSA)) || (!defined(LEFT) && defined(TRANSA))
- temp = bk-off;
- #elif defined(LEFT)
- temp = off+2; // number of values in A
- #else
- temp = off+2; // number of values in B
- #endif
-
- for (k=0; k<temp; k++)
- {
- b0_r = ptrbb[2*0+0]; b0_i = ptrbb[2*0+1];
- b1_r = ptrbb[2*1+0]; b1_i = ptrbb[2*1+1];
-
- a0_r = ptrba[2*0+0]; a0_i = ptrba[2*0+1];
- MADD(res00, a0, b0);
- MADD(res10, a0, b1);
-
- a1_r = ptrba[2*1+0]; a1_i = ptrba[2*1+1];
- MADD(res01, a1, b0);
- MADD(res11, a1, b1);
-
-
- ptrba = ptrba+4;
- ptrbb = ptrbb+4;
- }
-
- MADD_ALPHA_N_STORE(C0, res00, alpha);
- C0 = C0 + 2;
- MADD_ALPHA_N_STORE(C0, res01, alpha);
- C0 = C0 + 2;
-
- MADD_ALPHA_N_STORE(C1, res10, alpha);
- C1 = C1 + 2;
- MADD_ALPHA_N_STORE(C1, res11, alpha);
- C1 = C1 + 2;
-
- #if ( defined(LEFT) && defined(TRANSA)) || (!defined(LEFT) && !defined(TRANSA))
- temp = bk - off;
- #ifdef LEFT
- temp -= 2; // number of values in A
- #else
- temp -= 2; // number of values in B
- #endif
- ptrba += temp*2*2;
- ptrbb += temp*2*2;
- #endif
-
- #ifdef LEFT
- off += 2; // number of values in A
- #endif
-
- }
-
- if ( bm & 1 ) // do any 1x2 loop
- {
-
- #if (defined(LEFT) && defined(TRANSA)) || (!defined(LEFT) && !defined(TRANSA))
- ptrbb = bb;
- #else
- ptrba += off*1*2;
- ptrbb = bb + off*2*2;
- #endif
-
- res00_r = 0;
- res00_i = 0;
-
- res10_r = 0;
- res10_i = 0;
-
-
- #if (defined(LEFT) && !defined(TRANSA)) || (!defined(LEFT) && defined(TRANSA))
- temp = bk-off;
- #elif defined(LEFT)
- temp = off+1; // number of values in A
- #else
- temp = off+2; // number of values in B
- #endif
-
- for (k=0; k<temp; k++)
- {
- b0_r = ptrbb[2*0+0]; b0_i = ptrbb[2*0+1];
- b1_r = ptrbb[2*1+0]; b1_i = ptrbb[2*1+1];
-
- a0_r = ptrba[2*0+0]; a0_i = ptrba[2*0+1];
- MADD(res00, a0, b0);
- MADD(res10, a0, b1);
-
- ptrba = ptrba+2;
- ptrbb = ptrbb+4;
- }
-
- MADD_ALPHA_N_STORE(C0, res00, alpha);
- C0 = C0 + 2;
-
- MADD_ALPHA_N_STORE(C1, res10, alpha);
- C1 = C1 + 2;
-
- #if ( defined(LEFT) && defined(TRANSA)) || (!defined(LEFT) && !defined(TRANSA))
- temp = bk - off;
- #ifdef LEFT
- temp -= 1; // number of values in A
- #else
- temp -= 2; // number of values in B
- #endif
- ptrba += temp*1*2;
- ptrbb += temp*2*2;
- #endif
-
- #ifdef LEFT
- off += 1; // number of values in A
- #endif
-
- }
-
-
- #if defined(TRMMKERNEL) && !defined(LEFT)
- off += 2;
- #endif
-
- k = (bk<<2);
- bb = bb+k;
- i = (ldc<<2);
- C = C+i;
- }
-
-
-
-
-
-
-
- for (j=0; j<(bn&1); j+=1) // do the Mx1 loops
- {
- C0 = C;
-
- #if defined(TRMMKERNEL) && defined(LEFT)
- off = offset;
- #endif
-
- ptrba = ba;
-
- for (i=0; i<bm/4; i+=1) // do blocks of 4x1 loops
- {
-
- #if (defined(LEFT) && defined(TRANSA)) || (!defined(LEFT) && !defined(TRANSA))
- ptrbb = bb;
- #else
- ptrba += off*4*2;
- ptrbb = bb + off*1*2;
- #endif
-
- res00_r = 0;
- res00_i = 0;
- res01_r = 0;
- res01_i = 0;
- res02_r = 0;
- res02_i = 0;
- res03_r = 0;
- res03_i = 0;
-
-
- #if (defined(LEFT) && !defined(TRANSA)) || (!defined(LEFT) && defined(TRANSA))
- temp = bk-off;
- #elif defined(LEFT)
- temp = off+4; // number of values in A
- #else
- temp = off+1; // number of values in B
- #endif
-
- for (k=0; k<temp; k++)
- {
- b0_r = ptrbb[2*0+0]; b0_i = ptrbb[2*0+1];
-
- a0_r = ptrba[2*0+0]; a0_i = ptrba[2*0+1];
- MADD(res00, a0, b0);
-
- a1_r = ptrba[2*1+0]; a1_i = ptrba[2*1+1];
- MADD(res01, a1, b0);
-
- a0_r = ptrba[2*2+0]; a0_i = ptrba[2*2+1];
- MADD(res02, a0, b0);
-
- a1_r = ptrba[2*3+0]; a1_i = ptrba[2*3+1];
- MADD(res03, a1, b0);
-
- ptrba = ptrba+8;
- ptrbb = ptrbb+2;
- }
-
- MADD_ALPHA_N_STORE(C0, res00, alpha);
- C0 = C0 + 2;
- MADD_ALPHA_N_STORE(C0, res01, alpha);
- C0 = C0 + 2;
- MADD_ALPHA_N_STORE(C0, res02, alpha);
- C0 = C0 + 2;
- MADD_ALPHA_N_STORE(C0, res03, alpha);
- C0 = C0 + 2;
-
-
- #if ( defined(LEFT) && defined(TRANSA)) || (!defined(LEFT) && !defined(TRANSA))
- temp = bk - off;
- #ifdef LEFT
- temp -= 4; // number of values in A
- #else
- temp -= 1; // number of values in B
- #endif
- ptrba += temp*4*2;
- ptrbb += temp*1*2;
- #endif
-
- #ifdef LEFT
- off += 4; // number of values in A
- #endif
-
- }
-
- if ( bm & 2 ) // do any 2x1 loop
- {
-
- #if (defined(LEFT) && defined(TRANSA)) || (!defined(LEFT) && !defined(TRANSA))
- ptrbb = bb;
- #else
- ptrba += off*2*2;
- ptrbb = bb + off*1*2;
- #endif
-
- res00_r = 0;
- res00_i = 0;
- res01_r = 0;
- res01_i = 0;
-
- #if (defined(LEFT) && !defined(TRANSA)) || (!defined(LEFT) && defined(TRANSA))
- temp = bk-off;
- #elif defined(LEFT)
- temp = off+2; // number of values in A
- #else
- temp = off+1; // number of values in B
- #endif
-
- for (k=0; k<temp; k++)
- {
- b0_r = ptrbb[2*0+0]; b0_i = ptrbb[2*0+1];
-
- a0_r = ptrba[2*0+0]; a0_i = ptrba[2*0+1];
- MADD(res00, a0, b0);
-
- a1_r = ptrba[2*1+0]; a1_i = ptrba[2*1+1];
- MADD(res01, a1, b0);
-
-
- ptrba = ptrba+4;
- ptrbb = ptrbb+2;
- }
-
- MADD_ALPHA_N_STORE(C0, res00, alpha);
- C0 = C0 + 2;
- MADD_ALPHA_N_STORE(C0, res01, alpha);
- C0 = C0 + 2;
-
-
- #if ( defined(LEFT) && defined(TRANSA)) || (!defined(LEFT) && !defined(TRANSA))
- temp = bk - off;
- #ifdef LEFT
- temp -= 2; // number of values in A
- #else
- temp -= 1; // number of values in B
- #endif
- ptrba += temp*2*2;
- ptrbb += temp*1*2;
- #endif
-
- #ifdef LEFT
- off += 2; // number of values in A
- #endif
-
- }
-
- if ( bm & 1 ) // do any 1x1 loop
- {
-
- #if (defined(LEFT) && defined(TRANSA)) || (!defined(LEFT) && !defined(TRANSA))
- ptrbb = bb;
- #else
- ptrba += off*1*2;
- ptrbb = bb + off*1*2;
- #endif
-
- res00_r = 0;
- res00_i = 0;
-
-
- #if (defined(LEFT) && !defined(TRANSA)) || (!defined(LEFT) && defined(TRANSA))
- temp = bk-off;
- #elif defined(LEFT)
- temp = off+1; // number of values in A
- #else
- temp = off+1; // number of values in B
- #endif
-
- for (k=0; k<temp; k++)
- {
- b0_r = ptrbb[2*0+0]; b0_i = ptrbb[2*0+1];
-
- a0_r = ptrba[2*0+0]; a0_i = ptrba[2*0+1];
- MADD(res00, a0, b0);
-
- ptrba = ptrba+2;
- ptrbb = ptrbb+2;
- }
-
- MADD_ALPHA_N_STORE(C0, res00, alpha);
- C0 = C0 + 2;
-
- #if ( defined(LEFT) && defined(TRANSA)) || (!defined(LEFT) && !defined(TRANSA))
- temp = bk - off;
- #ifdef LEFT
- temp -= 1; // number of values in A
- #else
- temp -= 1; // number of values in B
- #endif
- ptrba += temp*1*2;
- ptrbb += temp*1*2;
- #endif
-
- #ifdef LEFT
- off += 1; // number of values in A
- #endif
-
- }
-
-
-
- #if defined(TRMMKERNEL) && !defined(LEFT)
- off += 1;
- #endif
-
- k = (bk<<1);
- bb = bb+k;
- i = (ldc<<1);
- C = C+i;
- }
- return 0;
- }
|