8#ifndef QDP_SCALARSITE_SSE_BLAS_H
9#define QDP_SCALARSITE_SSE_BLAS_H
51#define QDP_SCALARSITE_USE_EVALUATE
63#if defined(QDP_SCALARSITE_USE_EVALUATE)
91 int total_n_3vec = (s.
end()-s.
start()+1);
151 REAL32 ar = -( a.
elem().elem().elem().elem());
157 int total_n_3vec = (s.
end()-s.
start()+1);
223 const BN &mulNode =
static_cast<const BN&
> (rhs.expression().left());
236 int total_n_3vec = (s.
end()-s.
start()+1);
306 const BN &mulNode =
static_cast<const BN&
> (rhs.expression().right());
386 const BN &mulNode =
static_cast<const BN&
> (rhs.expression().left());
400 int total_n_3vec = (s.
end()-s.
start()+1);
465 const BN &mulNode =
static_cast<const BN&
> (rhs.expression().right());
480 int total_n_3vec = (s.
end()-s.
start()+1);
542 int total_n_3vec = (s.
end()-s.
start()+1);
598 REAL32 ar = -( a.
elem().elem().elem().elem());
605 int total_n_3vec = (s.
end()-s.
start()+1);
671 const BN &mulNode =
static_cast<const BN&
> (rhs.expression().left());
684 int total_n_3vec = (s.
end()-s.
start()+1);
751 const BN &mulNode =
static_cast<const BN&
> (rhs.expression().right());
764 int total_n_3vec = (s.
end()-s.
start()+1);
829 const BN &mulNode =
static_cast<const BN&
> (rhs.expression().left());
842 int total_n_3vec = (s.
end()-s.
start()+1);
907 const BN &mulNode =
static_cast<const BN&
> (rhs.expression().right());
922 int total_n_3vec = (s.
end()-s.
start()+1);
973 cout <<
"SSE: v+v " << endl;
984 int total_n_3vec = (s.
end()-s.
start()+1);
1035 cout <<
"SSE: v-v " << endl;
1046 int total_n_3vec = (s.
end()-s.
start()+1);
1098 cout <<
"SSE: v = a*v " << endl;
1109 int total_n_3vec = (s.
end()-s.
start()+1);
1157 cout <<
"SSE: v = v*a " << endl;
1169 int total_n_3vec = (s.
end()-s.
start()+1);
1230 int total_n_3vec = (s.
end()-s.
start()+1);
1288 int total_n_3vec = (s.
end()-s.
start()+1);
1345 int total_n_3vec = (s.
end()-s.
start()+1);
1405 int total_n_3vec = (s.
end()-s.
start()+1);
1470 const BN &mulNode1 =
static_cast<const BN&
> (rhs.expression().left());
1471 const BN &mulNode2 =
static_cast<const BN&
> (rhs.expression().right());
1490 int total_n_3vec = (s.
end()-s.
start()+1);
1564 const BN1 &mulNode1 =
static_cast<const BN1&
> (rhs.expression().left());
1565 const BN2 &mulNode2 =
static_cast<const BN2&
> (rhs.expression().right());
1585 int total_n_3vec = (s.
end()-s.
start()+1);
1661 const BN1 &mulNode1 =
static_cast<const BN1&
> (rhs.expression().left());
1664 const BN2 &mulNode2 =
static_cast<const BN2&
> (rhs.expression().right());
1687 int total_n_3vec = (s.
end()-s.
start()+1);
1755 const BN &mulNode1 =
static_cast<const BN&
> (rhs.expression().left());
1756 const BN &mulNode2 =
static_cast<const BN&
> (rhs.expression().right());
1775 int total_n_3vec = (s.
end()-s.
start()+1);
1844 const BN &mulNode1 =
static_cast<const BN&
> (rhs.expression().left());
1845 const BN &mulNode2 =
static_cast<const BN&
> (rhs.expression().right());
1864 int total_n_3vec = (s.
end()-s.
start()+1);
1941 const BN1 &mulNode1 =
static_cast<const BN1&
> (rhs.expression().left());
1942 const BN2 &mulNode2 =
static_cast<const BN2&
> (rhs.expression().right());
1962 int total_n_3vec = (s.
end()-s.
start()+1);
2039 const BN1 &mulNode1 =
static_cast<const BN1&
> (rhs.expression().left());
2042 const BN2 &mulNode2 =
static_cast<const BN2&
> (rhs.expression().right());
2064 int total_n_3vec = (s.
end()-s.
start()+1);
2133 const BN &mulNode1 =
static_cast<const BN&
> (rhs.expression().left());
2134 const BN &mulNode2 =
static_cast<const BN&
> (rhs.expression().right());
2153 int total_n_3vec = (s.
end()-s.
start()+1);
2215 int n_real = (s.
end() - s.
start() + 1);
2239 REAL32* s1ptr = (
REAL32 *)&(s1.elem(i).elem(0).elem(0).real());
2261 int n_real = (
all.end() -
all.start() + 1);
2263 arg.
vptr = (
REAL32*) &(s1.elem(
all.start()).elem(0).elem(0).real());
2300 unsigned long n_3vec = (
all.end() -
all.start() + 1)*
Ns;
2304 (
REAL32 *)&(v1.elem(
all.start()).elem(0).elem(0).real()),
2305 (
REAL32 *)&(v2.elem(
all.start()).elem(0).elem(0).real()),
2313 lprod.elem().elem().elem().real() = ip[0];
2314 lprod.elem().elem().elem().imag() = ip[1];
2342 unsigned long n_3vec = (s.
end() - s.
start() + 1)*
Ns;
2344 (
REAL32 *)&(v1.elem(s.
start()).elem(0).elem(0).real()),
2345 (
REAL32 *)&(v2.elem(s.
start()).elem(0).elem(0).real()),
2363 (
REAL32 *)&(v1.elem(i).elem(0).elem(0).real()),
2364 (
REAL32 *)&(v2.elem(i).elem(0).elem(0).real()),
2374 lprod.elem().elem().elem().real() = ip[0];
2375 lprod.elem().elem().elem().imag() = ip[1];
2390 QDPIO::cout <<
"BJ: innerProductReal all" << endl;
2400 unsigned long n_3vec = (
all.end() -
all.start() + 1)*
Ns;
2404 (
REAL32 *)&(v1.elem(
all.start()).elem(0).elem(0).real()),
2405 (
REAL32 *)&(v2.elem(
all.start()).elem(0).elem(0).real()),
2413 lprod.elem().elem().elem().elem() = ip_re;
2440 unsigned long n_3vec = (s.
end() - s.
start() + 1)*
Ns;
2442 (
REAL32 *)&(v1.elem(s.
start()).elem(0).elem(0).real()),
2443 (
REAL32 *)&(v2.elem(s.
start()).elem(0).elem(0).real()),
2457 (
REAL32 *)&(v1.elem(i).elem(0).elem(0).real()),
2458 (
REAL32 *)&(v2.elem(i).elem(0).elem(0).real()),
2469 lprod.elem().elem().elem().elem() = ip_re;
2479 QDPIO::cout <<
"Using SSE multi1d sumsq all" << endl;
2482 int n_real = (
all.end() -
all.start() + 1);
2484 for(
int n=0; n < s1.size(); ++n)
2486 const REAL32 *s1ptr = &(s1[n].elem(
all.start()).elem(0).elem(0).real());
2507 QDPIO::cout <<
"Using SSE multi1d sumsq all" << endl;
2513 int n_real = (s.
end() - s.
start() + 1);
2515 for(
int n=0; n < s1.size(); ++n) {
2517 const REAL32 *s1ptr = &(s1[n].elem(s.
start()).elem(0).elem(0).real());
2529 for(
int n=0; n < s1.size(); ++n) {
2533 const REAL32 *s1ptr = &(s1[n].elem(i).elem(0).elem(0).
real());
2553 QDPIO::cout <<
"BJ: multi1d innerProduct all" << endl;
2565 unsigned long n_3vec = (
all.end() -
all.start() + 1)*
Ns;
2567 for(
int n=0; n < v1.size(); ++n)
2575 (
REAL32 *)&(v1[n].elem(
all.start()).elem(0).elem(0).real()),
2576 (
REAL32 *)&(v2[n].elem(
all.start()).elem(0).elem(0).real()),
2587 lprod.elem().elem().elem().real() = ip[0];
2588 lprod.elem().elem().elem().imag() = ip[1];
2601 QDPIO::cout <<
"BJ: multi1d innerProduct subset" << endl;
2615 unsigned long n_3vec = (s.
end() - s.
start() + 1)*
Ns;
2617 for(
int n=0; n < v1.size(); ++n) {
2626 (
REAL32 *)&(v1[n].elem(s.
start()).elem(0).elem(0).real()),
2627 (
REAL32 *)&(v2[n].elem(s.
start()).elem(0).elem(0).real()),
2636 unsigned long n_3vec =
Ns;
2642 for(
int n=0; n < v1.size(); ++n) {
2653 (
REAL32 *)&(v1[n].elem(i).elem(0).elem(0).
real()),
2654 (
REAL32 *)&(v2[n].elem(i).elem(0).elem(0).
real()),
2666 lprod.elem().elem().elem().real() = ip[0];
2667 lprod.elem().elem().elem().imag() = ip[1];
2681 QDPIO::cout <<
"BJ: innerProductReal(multi1d) all" << endl;
2691 unsigned long n_3vec = (
all.end() -
all.start() + 1)*
Ns;
2693 for(
int n=0; n < v1.size(); ++n)
2699 (
REAL32 *)&(v1[n].elem(
all.start()).elem(0).elem(0).real()),
2700 (
REAL32 *)&(v2[n].elem(
all.start()).elem(0).elem(0).real()),
2711 lprod.elem().elem().elem().elem() = ip_re;
2727 QDPIO::cout <<
"BJ: innerProductReal(multi1d) all" << endl;
2738 unsigned long n_3vec = (s.
end() - s.
start() + 1)*
Ns;
2740 for(
int n=0; n < v1.size(); ++n) {
2746 (
REAL32 *)&(v1[n].elem(s.
start()).elem(0).elem(0).real()),
2747 (
REAL32 *)&(v2[n].elem(s.
start()).elem(0).elem(0).real()),
2754 unsigned long n_3vec =
Ns;
2757 for(
int n=0; n < v1.size(); ++n) {
2764 (
REAL32 *)&(v1[n].elem(i).elem(0).elem(0).
real()),
2765 (
REAL32 *)&(v2[n].elem(i).elem(0).elem(0).
real()),
2777 lprod.elem().elem().elem().elem() = ip_re;
2787#if defined(QDP_SCALARSITE_DEBUG)
2788#undef QDP_SCALARSITE_DEBUG
2791#if defined(QDP_SCALARSITE_USE_EVALUATE)
2792#undef QDP_SCALARSITE_USE_EVALUATE
Outer grid Scalar class */.
Primitive spin Vector class.
Expression class for QDP.
QDPType - major type class/container for all QDP objects.
Subsets - controls how lattices are looped.
const multi1d< int > & siteTable() const
bool hasOrderedRep() const
Container for a multi-dimensional 1D array.
const T * slice() const
Return ref to a column slice.
BinaryReturn< C1, C2, FnInnerProductReal >::Type_t innerProductReal(const QDPType< T1, C1 > &s1, const QDPType< T2, C2 > &s2)
OScalar = innerProductReal(adj(source1)*source2).
UnaryReturn< C, FnNorm2 >::Type_t norm2(const QDPType< T, C > &s1)
OScalar = norm2(trace(adj(source)*source)).
BinaryReturn< C1, C2, FnInnerProduct >::Type_t innerProduct(const QDPType< T1, C1 > &s1, const QDPType< T2, C2 > &s2)
OScalar = innerProduct(adj(source1)*source2).
void evaluate(OLattice< DCol > &d, const OpAssign &op, const QDPExpr< BinaryNode< OpMultiply, Reference< QDPType< DCol, OLattice< DCol > > >, Reference< QDPType< DCol, OLattice< DCol > > > >, OLattice< DCol > > &rhs, const Subset &s)
Subset all
Default all subset.
StandardOutputStream cout
void globalSum(T &dest)
Sum across all nodes.
void globalSumArray(unsigned int *dest, int len)
Wrapper to get a functional unsigned global sum.
Yet another random number generator.
void local_sumsq_24_48(REAL64 *Out, REAL32 *In, int n_3vec)
PScalar< PScalar< RScalar< REAL > > > TScal
MakeReturn< UnaryNode< FnReal, typenameCreateLeaf< QDPExpr< T1, C1 > >::Leaf_t >, typenameUnaryReturn< C1, FnReal >::Type_t >::Expression_t real(const QDPExpr< T1, C1 > &l)
void dispatch_to_threads(int numSiteTable, Arg a, void(*func)(int, int, int, Arg *))
void unordered_sse_vOp_y_evaluate_function(int lo, int hi, int myId, unordered_sse_vOp_y_user_arg *a)
void local_vcdot_real(REAL64 *Out_re, REAL32 *V1, REAL32 *V2, int n_3vec)
void unordered_sse_vaxOpy3_y_evaluate_function(int lo, int hi, int myId, unordered_sse_vaxOpy3_y_user_arg *a)
void local_vcdot(REAL64 *Out_re, REAL64 *Out_im, REAL32 *V1, REAL32 *V2, int n_3vec)
void unordered_sse_vscal_evaluate_function(int lo, int hi, int myId, unordered_sse_vscal_user_arg *a)
void vaxpby3(REAL *Out, REAL *ap, REAL *xp, REAL *bp, REAL *yp, int n_3vec)
void ordered_sse_vaxOpy3_evaluate_function(int lo, int hi, int myId, ordered_sse_vaxOpy3_user_arg *a)
void vsub(REAL *Out, REAL *In1, REAL *In2, int n_3vec)
void ordered_norm_single_func(int lo, int hi, int myId, ordered_sse_norm_single_user_arg *a)
void vaxpy3(REAL *Out, REAL *scalep, REAL *InScale, REAL *Add, int n_4vec)
void vadd(REAL *Out, REAL *In1, REAL *In2, int n_3vec)
void vaxmy3(REAL *Out, REAL *scalep, REAL *InScale, REAL *Sub, int n_3vec)
void vaxmby3(REAL *Out, REAL *ap, REAL *xp, REAL *bp, REAL *yp, int n_3vec)
void unordered_vOp_z_evaluate_function(int lo, int hi, int myId, unordered_sse_vOp_z_user_arg *a)
void ordered_sse_vOp_evaluate_function(int lo, int hi, int myId, ordered_sse_vOp_user_arg *a)
void vscal(REAL *Out, REAL *scalep, REAL *In, int n_3vec)
PSpinVector< PColorVector< RComplex< REAL >, 3 >, Ns > TVec
void unordered_sse_vaxOpby3_evaluate_function(int lo, int hi, int myId, unordered_sse_vaxOpby3_user_arg *arg)
void ordered_sse_vscal_evaluate_function(int lo, int hi, int myId, ordered_sse_vscal_user_arg *a)
void unordered_sse_vaxOpy3_z_evaluate_function(int lo, int hi, int myId, unordered_sse_vaxOpy3_z_user_arg *a)
void ordered_sse_vaxOpby3_evaluate_function(int lo, int hi, int myId, ordered_sse_vaxOpby3_user_arg *arg)
void unordered_sse_vaxOpy3_z_evaluate_function(int lo, int hi, int myId, unordered_sse_vaxOpy3_z_user_arg *a)
void ordered_sse_vaxOpy3_evaluate_function(int lo, int hi, int myId, ordered_sse_vaxOpy3_user_arg *a)
void(* func)(REAL64 *, REAL32 *, int)
const OLattice< TVec > & x
const OLattice< TVec > & y
const OLattice< TVec > & x