4
* -- SuperLU routine (version 2.0) --
5
* Univ. of California Berkeley, Xerox Palo Alto Research Center,
6
* and Lawrence Berkeley National Lab.
11
Copyright (c) 1994 by Xerox Corporation. All rights reserved.
13
THIS MATERIAL IS PROVIDED AS IS, WITH ABSOLUTELY NO WARRANTY
14
EXPRESSED OR IMPLIED. ANY USE IS AT YOUR OWN RISK.
16
Permission is hereby granted to use or copy this program for any
17
purpose, provided the above notices are retained on all copies.
18
Permission to modify the code and to distribute modified code is
19
granted, provided the above notices are retained, and a notice that
20
the code was modified is included with the above copyright notice.
28
* Performs numeric block updates within the relaxed snode.
32
const int jcol, /* in */
33
const int jsupno, /* in */
34
const int fsupc, /* in */
35
doublecomplex *dense, /* in */
36
doublecomplex *tempv, /* working array */
37
GlobalLU_t *Glu /* modified */
40
#ifdef USE_VENDOR_BLAS
42
_fcd ftcs1 = _cptofcd("L", strlen("L")),
43
ftcs2 = _cptofcd("N", strlen("N")),
44
ftcs3 = _cptofcd("U", strlen("U"));
46
int incx = 1, incy = 1;
47
doublecomplex alpha = {-1.0, 0.0}, beta = {1.0, 0.0};
50
doublecomplex comp_zero = {0.0, 0.0};
51
int luptr, nsupc, nsupr, nrow;
52
int isub, irow, i, iptr;
53
register int ufirst, nextlu;
57
extern SuperLUStat_t SuperLUStat;
58
flops_t *ops = SuperLUStat.ops;
65
nextlu = xlusup[jcol];
68
* Process the supernodal portion of L\U[*,j]
70
for (isub = xlsub[fsupc]; isub < xlsub[fsupc+1]; isub++) {
72
lusup[nextlu] = dense[irow];
73
dense[irow] = comp_zero;
77
xlusup[jcol + 1] = nextlu; /* Initialize xlusup for next column */
81
luptr = xlusup[fsupc];
82
nsupr = xlsub[fsupc+1] - xlsub[fsupc];
83
nsupc = jcol - fsupc; /* Excluding jcol */
84
ufirst = xlusup[jcol]; /* Points to the beginning of column
85
jcol in supernode L\U(jsupno). */
88
ops[TRSV] += 4 * nsupc * (nsupc - 1);
89
ops[GEMV] += 8 * nrow * nsupc;
91
#ifdef USE_VENDOR_BLAS
93
CTRSV( ftcs1, ftcs2, ftcs3, &nsupc, &lusup[luptr], &nsupr,
94
&lusup[ufirst], &incx );
95
CGEMV( ftcs2, &nrow, &nsupc, &alpha, &lusup[luptr+nsupc], &nsupr,
96
&lusup[ufirst], &incx, &beta, &lusup[ufirst+nsupc], &incy );
98
ztrsv_( "L", "N", "U", &nsupc, &lusup[luptr], &nsupr,
99
&lusup[ufirst], &incx );
100
zgemv_( "N", &nrow, &nsupc, &alpha, &lusup[luptr+nsupc], &nsupr,
101
&lusup[ufirst], &incx, &beta, &lusup[ufirst+nsupc], &incy );
104
zlsolve ( nsupr, nsupc, &lusup[luptr], &lusup[ufirst] );
105
zmatvec ( nsupr, nrow, nsupc, &lusup[luptr+nsupc],
106
&lusup[ufirst], &tempv[0] );
108
/* Scatter tempv[*] into lusup[*] */
109
iptr = ufirst + nsupc;
110
for (i = 0; i < nrow; i++) {
111
z_sub(&lusup[iptr], &lusup[iptr], &tempv[i]);
113
tempv[i] = comp_zero;