Actual source code: mpimatmatmult.c

  1: /*
  2:   Defines matrix-matrix product routines for pairs of MPIAIJ matrices
  3:           C = A * B
  4: */
  5: #include <../src/mat/impls/aij/seq/aij.h>
  6: #include <../src/mat/utils/freespace.h>
  7: #include <../src/mat/impls/aij/mpi/mpiaij.h>
  8: #include <petscbt.h>
  9: #include <../src/mat/impls/dense/mpi/mpidense.h>
 10: #include <petsc/private/vecimpl.h>
 11: #include <petsc/private/sfimpl.h>

 13: #if PetscDefined(HAVE_HYPRE)
 14: PETSC_INTERN PetscErrorCode MatMatMultSymbolic_AIJ_AIJ_wHYPRE(Mat, Mat, PetscReal, Mat);
 15: #endif

 17: PETSC_INTERN PetscErrorCode MatProductSymbolic_ABt_MPIAIJ_MPIAIJ(Mat C)
 18: {
 19:   Mat_Product *product = C->product;
 20:   Mat          B       = product->B;

 22:   PetscFunctionBegin;
 23:   PetscCall(MatTranspose(B, MAT_INITIAL_MATRIX, &product->B));
 24:   PetscCall(MatDestroy(&B));
 25:   PetscCall(MatProductSymbolic_AB_MPIAIJ_MPIAIJ(C));
 26:   PetscFunctionReturn(PETSC_SUCCESS);
 27: }

 29: PETSC_INTERN PetscErrorCode MatProductSymbolic_AB_MPIAIJ_MPIAIJ(Mat C)
 30: {
 31:   Mat_Product        *product = C->product;
 32:   Mat                 A = product->A, B = product->B;
 33:   MatProductAlgorithm alg  = product->alg;
 34:   PetscReal           fill = product->fill;
 35:   PetscBool           flg;

 37:   PetscFunctionBegin;
 38:   /* scalable */
 39:   PetscCall(PetscStrcmp(alg, "scalable", &flg));
 40:   if (flg || C->structure_only) {
 41:     if (C->structure_only) PetscCall(MatProductSetAlgorithm(C, "scalable"));
 42:     PetscCall(MatMatMultSymbolic_MPIAIJ_MPIAIJ(A, B, fill, C));
 43:     PetscFunctionReturn(PETSC_SUCCESS);
 44:   }

 46:   /* nonscalable */
 47:   PetscCall(PetscStrcmp(alg, "nonscalable", &flg));
 48:   if (flg) {
 49:     PetscCall(MatMatMultSymbolic_MPIAIJ_MPIAIJ_nonscalable(A, B, fill, C));
 50:     PetscFunctionReturn(PETSC_SUCCESS);
 51:   }

 53:   /* seqmpi */
 54:   PetscCall(PetscStrcmp(alg, "seqmpi", &flg));
 55:   if (flg) {
 56:     PetscCall(MatMatMultSymbolic_MPIAIJ_MPIAIJ_seqMPI(A, B, fill, C));
 57:     PetscFunctionReturn(PETSC_SUCCESS);
 58:   }

 60:   /* backend general code */
 61:   PetscCall(PetscStrcmp(alg, "backend", &flg));
 62:   if (flg) {
 63:     PetscCall(MatProductSymbolic_MPIAIJBACKEND(C));
 64:     PetscFunctionReturn(PETSC_SUCCESS);
 65:   }

 67: #if PetscDefined(HAVE_HYPRE)
 68:   PetscCall(PetscStrcmp(alg, "hypre", &flg));
 69:   if (flg) {
 70:     PetscCall(MatMatMultSymbolic_AIJ_AIJ_wHYPRE(A, B, fill, C));
 71:     PetscFunctionReturn(PETSC_SUCCESS);
 72:   }
 73: #endif
 74:   SETERRQ(PetscObjectComm((PetscObject)C), PETSC_ERR_SUP, "Mat Product Algorithm is not supported");
 75: }

 77: PetscErrorCode MatProductCtxDestroy_MPIAIJ_MatMatMult(PetscCtxRt data)
 78: {
 79:   MatProductCtx_APMPI *ptap = *(MatProductCtx_APMPI **)data;

 81:   PetscFunctionBegin;
 82:   PetscCall(PetscFree2(ptap->startsj_s, ptap->startsj_r));
 83:   PetscCall(PetscFree(ptap->bufa));
 84:   PetscCall(MatDestroy(&ptap->P_loc));
 85:   PetscCall(MatDestroy(&ptap->P_oth));
 86:   PetscCall(MatDestroy(&ptap->Pt));
 87:   PetscCall(PetscFree(ptap->api));
 88:   PetscCall(PetscFree(ptap->apj));
 89:   PetscCall(PetscFree(ptap->apa));
 90:   PetscCall(PetscFree(ptap));
 91:   PetscFunctionReturn(PETSC_SUCCESS);
 92: }

 94: PetscErrorCode MatMatMultNumeric_MPIAIJ_MPIAIJ_nonscalable(Mat A, Mat P, Mat C)
 95: {
 96:   Mat_MPIAIJ          *a = (Mat_MPIAIJ *)A->data, *c = (Mat_MPIAIJ *)C->data;
 97:   Mat_SeqAIJ          *ad = (Mat_SeqAIJ *)a->A->data, *ao = (Mat_SeqAIJ *)a->B->data;
 98:   Mat_SeqAIJ          *cd = (Mat_SeqAIJ *)c->A->data, *co = (Mat_SeqAIJ *)c->B->data;
 99:   PetscScalar         *cda, *coa;
100:   Mat_SeqAIJ          *p_loc, *p_oth;
101:   PetscScalar         *apa, *ca;
102:   PetscInt             cm = C->rmap->n;
103:   MatProductCtx_APMPI *ptap;
104:   PetscInt            *api, *apj, *apJ, i, k;
105:   PetscInt             cstart = C->cmap->rstart;
106:   PetscInt             cdnz, conz, k0, k1;
107:   const PetscScalar   *dummy1, *dummy2, *dummy3, *dummy4;
108:   MPI_Comm             comm;
109:   PetscMPIInt          size;

111:   PetscFunctionBegin;
112:   MatCheckProduct(C, 3);
113:   ptap = (MatProductCtx_APMPI *)C->product->data;
114:   PetscCheck(ptap, PetscObjectComm((PetscObject)C), PETSC_ERR_ARG_WRONGSTATE, "PtAP cannot be computed. Missing data");
115:   PetscCall(PetscObjectGetComm((PetscObject)A, &comm));
116:   PetscCallMPI(MPI_Comm_size(comm, &size));
117:   PetscCheck(ptap->P_oth || size <= 1, PetscObjectComm((PetscObject)C), PETSC_ERR_ARG_WRONGSTATE, "AP cannot be reused. Do not call MatProductClear()");

119:   /* flag CPU mask for C */
120: #if PetscDefined(HAVE_DEVICE)
121:   if (C->offloadmask != PETSC_OFFLOAD_UNALLOCATED) C->offloadmask = PETSC_OFFLOAD_CPU;
122:   if (c->A->offloadmask != PETSC_OFFLOAD_UNALLOCATED) c->A->offloadmask = PETSC_OFFLOAD_CPU;
123:   if (c->B->offloadmask != PETSC_OFFLOAD_UNALLOCATED) c->B->offloadmask = PETSC_OFFLOAD_CPU;
124: #endif

126:   /* 1) get P_oth = ptap->P_oth  and P_loc = ptap->P_loc */
127:   /* update numerical values of P_oth and P_loc */
128:   PetscCall(MatGetBrowsOfAoCols_MPIAIJ(A, P, MAT_REUSE_MATRIX, &ptap->startsj_s, &ptap->startsj_r, &ptap->bufa, &ptap->P_oth));
129:   PetscCall(MatMPIAIJGetLocalMat(P, MAT_REUSE_MATRIX, &ptap->P_loc));

131:   /* 2) compute numeric C_loc = A_loc*P = Ad*P_loc + Ao*P_oth */
132:   /* get data from symbolic products */
133:   p_loc = (Mat_SeqAIJ *)ptap->P_loc->data;
134:   p_oth = NULL;
135:   if (size > 1) p_oth = (Mat_SeqAIJ *)ptap->P_oth->data;

137:   /* get apa for storing dense row A[i,:]*P */
138:   apa = ptap->apa;

140:   api = ptap->api;
141:   apj = ptap->apj;
142:   /* trigger copy to CPU */
143:   PetscCall(MatSeqAIJGetArrayRead(a->A, &dummy1));
144:   PetscCall(MatSeqAIJGetArrayRead(a->B, &dummy2));
145:   PetscCall(MatSeqAIJGetArrayRead(ptap->P_loc, &dummy3));
146:   if (ptap->P_oth) PetscCall(MatSeqAIJGetArrayRead(ptap->P_oth, &dummy4));
147:   PetscCall(MatSeqAIJGetArrayWrite(c->A, &cda));
148:   PetscCall(MatSeqAIJGetArrayWrite(c->B, &coa));
149:   for (i = 0; i < cm; i++) {
150:     /* compute apa = A[i,:]*P */
151:     AProw_nonscalable(i, ad, ao, p_loc, p_oth, apa);

153:     /* set values in C */
154:     apJ  = PetscSafePointerPlusOffset(apj, api[i]);
155:     cdnz = cd->i[i + 1] - cd->i[i];
156:     conz = co->i[i + 1] - co->i[i];

158:     /* 1st off-diagonal part of C */
159:     ca = PetscSafePointerPlusOffset(coa, co->i[i]);
160:     k  = 0;
161:     for (k0 = 0; k0 < conz; k0++) {
162:       if (apJ[k] >= cstart) break;
163:       ca[k0]        = apa[apJ[k]];
164:       apa[apJ[k++]] = 0.0;
165:     }

167:     /* diagonal part of C */
168:     ca = PetscSafePointerPlusOffset(cda, cd->i[i]);
169:     for (k1 = 0; k1 < cdnz; k1++) {
170:       ca[k1]        = apa[apJ[k]];
171:       apa[apJ[k++]] = 0.0;
172:     }

174:     /* 2nd off-diagonal part of C */
175:     ca = PetscSafePointerPlusOffset(coa, co->i[i]);
176:     for (; k0 < conz; k0++) {
177:       ca[k0]        = apa[apJ[k]];
178:       apa[apJ[k++]] = 0.0;
179:     }
180:   }
181:   PetscCall(MatSeqAIJRestoreArrayRead(a->A, &dummy1));
182:   PetscCall(MatSeqAIJRestoreArrayRead(a->B, &dummy2));
183:   PetscCall(MatSeqAIJRestoreArrayRead(ptap->P_loc, &dummy3));
184:   if (ptap->P_oth) PetscCall(MatSeqAIJRestoreArrayRead(ptap->P_oth, &dummy4));
185:   PetscCall(MatSeqAIJRestoreArrayWrite(c->A, &cda));
186:   PetscCall(MatSeqAIJRestoreArrayWrite(c->B, &coa));

188:   PetscCall(MatAssemblyBegin(C, MAT_FINAL_ASSEMBLY));
189:   PetscCall(MatAssemblyEnd(C, MAT_FINAL_ASSEMBLY));
190:   PetscFunctionReturn(PETSC_SUCCESS);
191: }

193: PetscErrorCode MatMatMultSymbolic_MPIAIJ_MPIAIJ_nonscalable(Mat A, Mat P, PetscReal fill, Mat C)
194: {
195:   MPI_Comm             comm;
196:   PetscMPIInt          size;
197:   MatProductCtx_APMPI *ptap;
198:   PetscFreeSpaceList   free_space = NULL, current_space = NULL;
199:   Mat_MPIAIJ          *a  = (Mat_MPIAIJ *)A->data;
200:   Mat_SeqAIJ          *ad = (Mat_SeqAIJ *)a->A->data, *ao = (Mat_SeqAIJ *)a->B->data, *p_loc, *p_oth;
201:   PetscInt            *pi_loc, *pj_loc, *pi_oth, *pj_oth, *dnz, *onz;
202:   PetscInt            *adi = ad->i, *adj = ad->j, *aoi = ao->i, *aoj = ao->j, rstart = A->rmap->rstart;
203:   PetscInt            *lnk, i, pnz, row, *api, *apj, *Jptr, apnz, nspacedouble = 0, j, nzi;
204:   PetscInt             am = A->rmap->n, pN = P->cmap->N, pn = P->cmap->n, pm = P->rmap->n;
205:   PetscBT              lnkbt;
206:   PetscReal            afill;
207:   MatType              mtype;

209:   PetscFunctionBegin;
210:   MatCheckProduct(C, 4);
211:   PetscCheck(!C->product->data, PETSC_COMM_SELF, PETSC_ERR_PLIB, "Extra product struct not empty");
212:   PetscCall(PetscObjectGetComm((PetscObject)A, &comm));
213:   PetscCallMPI(MPI_Comm_size(comm, &size));

215:   /* create struct MatProductCtx_APMPI and attached it to C later */
216:   PetscCall(PetscNew(&ptap));

218:   /* get P_oth by taking rows of P (= non-zero cols of local A) from other processors */
219:   PetscCall(MatGetBrowsOfAoCols_MPIAIJ(A, P, MAT_INITIAL_MATRIX, &ptap->startsj_s, &ptap->startsj_r, &ptap->bufa, &ptap->P_oth));

221:   /* get P_loc by taking all local rows of P */
222:   PetscCall(MatMPIAIJGetLocalMat(P, MAT_INITIAL_MATRIX, &ptap->P_loc));

224:   p_loc  = (Mat_SeqAIJ *)ptap->P_loc->data;
225:   pi_loc = p_loc->i;
226:   pj_loc = p_loc->j;
227:   if (size > 1) {
228:     p_oth  = (Mat_SeqAIJ *)ptap->P_oth->data;
229:     pi_oth = p_oth->i;
230:     pj_oth = p_oth->j;
231:   } else {
232:     p_oth  = NULL;
233:     pi_oth = NULL;
234:     pj_oth = NULL;
235:   }

237:   /* first, compute symbolic AP = A_loc*P = A_diag*P_loc + A_off*P_oth */
238:   PetscCall(PetscMalloc1(am + 1, &api));
239:   ptap->api = api;
240:   api[0]    = 0;

242:   /* create and initialize a linked list */
243:   PetscCall(PetscLLCondensedCreate(pN, pN, &lnk, &lnkbt));

245:   /* Initial FreeSpace size is fill*(nnz(A)+nnz(P)) */
246:   PetscCall(PetscFreeSpaceGet(PetscRealIntMultTruncate(fill, PetscIntSumTruncate(adi[am], PetscIntSumTruncate(aoi[am], pi_loc[pm]))), &free_space));
247:   current_space = free_space;

249:   MatPreallocateBegin(comm, am, pn, dnz, onz);
250:   for (i = 0; i < am; i++) {
251:     /* diagonal portion of A */
252:     nzi = adi[i + 1] - adi[i];
253:     for (j = 0; j < nzi; j++) {
254:       row  = *adj++;
255:       pnz  = pi_loc[row + 1] - pi_loc[row];
256:       Jptr = pj_loc + pi_loc[row];
257:       /* add non-zero cols of P into the sorted linked list lnk */
258:       PetscCall(PetscLLCondensedAddSorted(pnz, Jptr, lnk, lnkbt));
259:     }
260:     /* off-diagonal portion of A */
261:     nzi = aoi[i + 1] - aoi[i];
262:     for (j = 0; j < nzi; j++) {
263:       row  = *aoj++;
264:       pnz  = pi_oth[row + 1] - pi_oth[row];
265:       Jptr = pj_oth + pi_oth[row];
266:       PetscCall(PetscLLCondensedAddSorted(pnz, Jptr, lnk, lnkbt));
267:     }
268:     /* add possible missing diagonal entry */
269:     if (C->force_diagonals) {
270:       j = i + rstart; /* column index */
271:       PetscCall(PetscLLCondensedAddSorted(1, &j, lnk, lnkbt));
272:     }

274:     apnz       = lnk[0];
275:     api[i + 1] = api[i] + apnz;

277:     /* if free space is not available, double the total space in the list */
278:     if (current_space->local_remaining < apnz) {
279:       PetscCall(PetscFreeSpaceGet(PetscIntSumTruncate(apnz, current_space->total_array_size), &current_space));
280:       nspacedouble++;
281:     }

283:     /* Copy data into free space, then initialize lnk */
284:     PetscCall(PetscLLCondensedClean(pN, apnz, current_space->array, lnk, lnkbt));
285:     PetscCall(MatPreallocateSet(i + rstart, apnz, current_space->array, dnz, onz));

287:     current_space->array += apnz;
288:     current_space->local_used += apnz;
289:     current_space->local_remaining -= apnz;
290:   }

292:   /* Allocate space for apj, initialize apj, and */
293:   /* destroy list of free space and other temporary array(s) */
294:   PetscCall(PetscMalloc1(api[am], &ptap->apj));
295:   apj = ptap->apj;
296:   PetscCall(PetscFreeSpaceContiguous(&free_space, ptap->apj));
297:   PetscCall(PetscLLDestroy(lnk, lnkbt));

299:   /* malloc apa to store dense row A[i,:]*P */
300:   PetscCall(PetscCalloc1(pN, &ptap->apa));

302:   /* set and assemble symbolic parallel matrix C */
303:   PetscCall(MatSetSizes(C, am, pn, PETSC_DETERMINE, PETSC_DETERMINE));
304:   PetscCall(MatSetBlockSizesFromMats(C, A, P));

306:   PetscCall(MatGetType(A, &mtype));
307:   PetscCall(MatSetType(C, mtype));
308:   PetscCall(MatMPIAIJSetPreallocation(C, 0, dnz, 0, onz));
309:   MatPreallocateEnd(dnz, onz);

311:   PetscCall(MatSetValues_MPIAIJ_CopyFromCSRFormat_Symbolic(C, apj, api));
312:   PetscCall(MatSetOption(C, MAT_NO_OFF_PROC_ENTRIES, PETSC_TRUE));
313:   PetscCall(MatAssemblyBegin(C, MAT_FINAL_ASSEMBLY));
314:   PetscCall(MatAssemblyEnd(C, MAT_FINAL_ASSEMBLY));
315:   PetscCall(MatSetOption(C, MAT_NEW_NONZERO_LOCATION_ERR, PETSC_TRUE));

317:   C->ops->matmultnumeric = MatMatMultNumeric_MPIAIJ_MPIAIJ_nonscalable;
318:   C->ops->productnumeric = MatProductNumeric_AB;

320:   /* attach the supporting struct to C for reuse */
321:   C->product->data    = ptap;
322:   C->product->destroy = MatProductCtxDestroy_MPIAIJ_MatMatMult;

324:   /* set MatInfo */
325:   afill = (PetscReal)api[am] / (adi[am] + aoi[am] + pi_loc[pm] + 1) + 1.e-5;
326:   if (afill < 1.0) afill = 1.0;
327:   C->info.mallocs           = nspacedouble;
328:   C->info.fill_ratio_given  = fill;
329:   C->info.fill_ratio_needed = afill;

331:   if (PetscDefined(USE_INFO)) {
332:     if (api[am]) {
333:       PetscCall(PetscInfo(C, "Reallocs %" PetscInt_FMT "; Fill ratio: given %g needed %g.\n", nspacedouble, (double)fill, (double)afill));
334:       PetscCall(PetscInfo(C, "Use MatMatMult(A,B,MatReuse,%g,&C) for best performance.;\n", (double)afill));
335:     } else PetscCall(PetscInfo(C, "Empty matrix product\n"));
336:   }
337:   PetscFunctionReturn(PETSC_SUCCESS);
338: }

340: static PetscErrorCode MatMatMultSymbolic_MPIAIJ_MPIDense(Mat, Mat, PetscReal, Mat);
341: static PetscErrorCode MatMatMultNumeric_MPIAIJ_MPIDense(Mat, Mat, Mat);

343: static PetscErrorCode MatProductSetFromOptions_MPIAIJ_MPIDense_AB(Mat C)
344: {
345:   Mat_Product *product = C->product;
346:   Mat          A = product->A, B = product->B;

348:   PetscFunctionBegin;
349:   if (A->cmap->rstart != B->rmap->rstart || A->cmap->rend != B->rmap->rend)
350:     SETERRQ(PETSC_COMM_SELF, PETSC_ERR_ARG_SIZ, "Matrix local dimensions are incompatible, (%" PetscInt_FMT ", %" PetscInt_FMT ") != (%" PetscInt_FMT ",%" PetscInt_FMT ")", A->cmap->rstart, A->cmap->rend, B->rmap->rstart, B->rmap->rend);

352:   C->ops->matmultsymbolic = MatMatMultSymbolic_MPIAIJ_MPIDense;
353:   C->ops->productsymbolic = MatProductSymbolic_AB;
354:   PetscFunctionReturn(PETSC_SUCCESS);
355: }

357: static PetscErrorCode MatProductSetFromOptions_MPIAIJ_MPIDense_AtB(Mat C)
358: {
359:   Mat_Product *product = C->product;
360:   Mat          A = product->A, B = product->B;

362:   PetscFunctionBegin;
363:   if (A->rmap->rstart != B->rmap->rstart || A->rmap->rend != B->rmap->rend)
364:     SETERRQ(PETSC_COMM_SELF, PETSC_ERR_ARG_SIZ, "Matrix local dimensions are incompatible, (%" PetscInt_FMT ", %" PetscInt_FMT ") != (%" PetscInt_FMT ",%" PetscInt_FMT ")", A->rmap->rstart, A->rmap->rend, B->rmap->rstart, B->rmap->rend);

366:   C->ops->transposematmultsymbolic = MatTransposeMatMultSymbolic_MPIAIJ_MPIDense;
367:   C->ops->productsymbolic          = MatProductSymbolic_AtB;
368:   PetscFunctionReturn(PETSC_SUCCESS);
369: }

371: PETSC_INTERN PetscErrorCode MatProductSetFromOptions_MPIAIJ_MPIDense(Mat C)
372: {
373:   Mat_Product *product = C->product;

375:   PetscFunctionBegin;
376:   switch (product->type) {
377:   case MATPRODUCT_AB:
378:     PetscCall(MatProductSetFromOptions_MPIAIJ_MPIDense_AB(C));
379:     break;
380:   case MATPRODUCT_AtB:
381:     PetscCall(MatProductSetFromOptions_MPIAIJ_MPIDense_AtB(C));
382:     break;
383:   default:
384:     break;
385:   }
386:   PetscFunctionReturn(PETSC_SUCCESS);
387: }

389: PETSC_INTERN PetscErrorCode MatMPIDenseScatterDestroy_Private(MPIAIJ_MPIDense *contents)
390: {
391:   PetscFunctionBegin;
392:   PetscCall(MatDestroy(&contents->workB));
393:   PetscCall(PetscSFDestroy(&contents->sf[0]));
394:   PetscCall(PetscSFDestroy(&contents->sf[1]));
395:   for (PetscInt i = 0; i < contents->nsends; i++) PetscCallMPI(MPI_Type_free(&contents->stype[i]));
396:   for (PetscInt i = 0; i < contents->nrecvs; i++) PetscCallMPI(MPI_Type_free(&contents->rtype[i]));
397:   PetscCall(PetscFree4(contents->stype, contents->rtype, contents->rwaits, contents->swaits));
398:   PetscFunctionReturn(PETSC_SUCCESS);
399: }

401: typedef struct {
402:   MPIAIJ_MPIDense scatter;
403:   Mat             workC; /* off-diagonal contribution A_o * workB on the device route, NULL otherwise */
404: } MPIAIJ_MPIDense_AB;

406: static PetscErrorCode MatMPIAIJ_MPIDenseDestroy(PetscCtxRt ctx)
407: {
408:   MPIAIJ_MPIDense_AB *data = *(MPIAIJ_MPIDense_AB **)ctx;

410:   PetscFunctionBegin;
411:   PetscCall(MatDestroy(&data->workC));
412:   PetscCall(MatMPIDenseScatterDestroy_Private(&data->scatter));
413:   PetscCall(PetscFree(data));
414:   PetscFunctionReturn(PETSC_SUCCESS);
415: }

417: PETSC_INTERN PetscErrorCode MatMPIDenseScatterSetUp_Private(VecScatter ctx, PetscInt nrows, PetscInt bs, PetscInt Am, Mat B, Mat C, MPIAIJ_MPIDense *contents, PetscInt *batchSize, PetscInt *numBatches)
418: {
419:   PetscInt        Bm = B->rmap->n, BN = B->cmap->N, Bbn, Bbs, numBb, ncols;
420:   MPI_Comm        comm;
421:   MPI_Datatype    type1;
422:   const PetscInt *sindices, *sstarts, *rstarts;
423:   PetscMPIInt    *disp;
424:   PetscMPIInt     nsends, nrecvs, nrows_to, nrows_from, bs_mpi;
425:   PetscBool       flg;

427:   PetscFunctionBegin;
428:   PetscCall(PetscObjectGetComm((PetscObject)C, &comm));
429:   PetscCall(MatDenseGetLDA(B, &contents->blda));

431:   /* Create column block of B and C for memory scalability when BN is too large */
432:   /* Estimate Bbn, column size of Bb */
433:   if (nrows) {
434:     Bbn = 2 * Am * BN / nrows;
435:     if (!Bbn) Bbn = 1;
436:   } else Bbn = BN;
437:   Bbs = B->cmap->bs;
438:   Bbn = Bbn / Bbs * Bbs;
439:   if (Bbn > BN) Bbn = BN;
440:   PetscCallMPI(MPIU_Allreduce(MPI_IN_PLACE, &Bbn, 1, MPIU_INT, MPI_MAX, comm));

442:   /* Enable runtime option for Bbn */
443:   PetscOptionsBegin(comm, ((PetscObject)C)->prefix, "MatProduct", "Mat");
444:   PetscCall(PetscOptionsDeprecated("-matmatmult_Bbn", "-matproduct_batch_size", "3.25", NULL));
445:   PetscCall(PetscOptionsBoundedInt("-matproduct_batch_size", "Number of dense columns per batch", "MatProduct", Bbn, &Bbn, NULL, 0));
446:   PetscOptionsEnd();
447:   Bbn = PetscMin(Bbn, BN);

449:   if (Bbn > 0 && Bbn < BN) numBb = BN / Bbn;
450:   else numBb = 0;
451:   if (numBb) PetscCall(PetscInfo(C, "Using column batches of size %" PetscInt_FMT " for %" PetscInt_FMT " dense columns\n", Bbn, BN));
452:   ncols = Bbn ? Bbn : BN;

454:   /* Create work matrix used to store off processor rows of B needed for local product, with the same type as the local block of B */
455:   PetscCall(MatCreate(PETSC_COMM_SELF, &contents->workB));
456:   PetscCall(MatSetSizes(contents->workB, nrows, ncols, nrows, ncols));
457:   PetscCall(MatSetType(contents->workB, ((PetscObject)((Mat_MPIDense *)B->data)->A)->type_name));
458:   PetscCall(MatSetUp(contents->workB));

460:   PetscCall(PetscObjectTypeCompare((PetscObject)((Mat_MPIDense *)B->data)->A, MATSEQDENSE, &flg));
461:   contents->ondevice = (PetscBool)!flg;
462:   if (contents->ondevice) {
463:     /* Use strided PetscSFs, which support device memory, instead of the MPI derived data types below */
464:     PetscCheck(ctx->vscat.bs <= 1 || contents->blda % ctx->vscat.bs == 0, PETSC_COMM_SELF, PETSC_ERR_ARG_INCOMP, "Leading dimension %" PetscInt_FMT " of the dense matrix must be a multiple of the block size %" PetscInt_FMT, contents->blda, ctx->vscat.bs);
465:     contents->ncols[0] = ncols;
466:     PetscCall(PetscSFCreateStridedSF(ctx, ncols, contents->blda, nrows, &contents->sf[0]));
467:     if (numBb && BN % Bbn) {
468:       contents->ncols[1] = BN % Bbn;
469:       PetscCall(PetscSFCreateStridedSF(ctx, BN % Bbn, contents->blda, nrows, &contents->sf[1]));
470:     }
471:   } else {
472:     PetscCall(VecScatterGetRemote_Private(ctx, PETSC_TRUE, &nsends, &sstarts, &sindices, NULL, NULL));
473:     PetscCall(VecScatterGetRemoteOrdered_Private(ctx, PETSC_FALSE, &nrecvs, &rstarts, NULL, NULL, NULL));

475:     /* Use MPI derived data type to reduce memory required by the send/recv buffers */
476:     PetscCall(PetscMalloc4(nsends, &contents->stype, nrecvs, &contents->rtype, nrecvs, &contents->rwaits, nsends, &contents->swaits));
477:     contents->nsends = nsends;
478:     contents->nrecvs = nrecvs;

480:     PetscCall(PetscMalloc1(PetscMax(Bm, 1), &disp));
481:     PetscCall(PetscMPIIntCast(bs, &bs_mpi));
482:     for (PetscMPIInt i = 0; i < nsends; i++) {
483:       PetscCall(PetscMPIIntCast(sstarts[i + 1] - sstarts[i], &nrows_to));
484:       for (PetscInt j = 0; j < nrows_to; j++) PetscCall(PetscMPIIntCast(sindices[sstarts[i] + j] * bs, &disp[j]));
485:       PetscCallMPI(MPI_Type_create_indexed_block(nrows_to, bs_mpi, disp, MPIU_SCALAR, &type1));
486:       PetscCallMPI(MPI_Type_create_resized(type1, 0, contents->blda * sizeof(PetscScalar), &contents->stype[i]));
487:       PetscCallMPI(MPI_Type_commit(&contents->stype[i]));
488:       PetscCallMPI(MPI_Type_free(&type1));
489:     }

491:     for (PetscMPIInt i = 0; i < nrecvs; i++) {
492:       /* received values from a process form a (nrows_from x Bbn) row block in workB (column-wise) */
493:       PetscCall(PetscMPIIntCast((rstarts[i + 1] - rstarts[i]) * bs, &nrows_from));
494:       disp[0] = 0;
495:       PetscCallMPI(MPI_Type_create_indexed_block(1, nrows_from, disp, MPIU_SCALAR, &type1));
496:       PetscCallMPI(MPI_Type_create_resized(type1, 0, nrows * sizeof(PetscScalar), &contents->rtype[i]));
497:       PetscCallMPI(MPI_Type_commit(&contents->rtype[i]));
498:       PetscCallMPI(MPI_Type_free(&type1));
499:     }

501:     PetscCall(PetscFree(disp));
502:     PetscCall(VecScatterRestoreRemote_Private(ctx, PETSC_TRUE /*send*/, &nsends, &sstarts, &sindices, NULL, NULL));
503:     PetscCall(VecScatterRestoreRemoteOrdered_Private(ctx, PETSC_FALSE /*recv*/, &nrecvs, &rstarts, NULL, NULL, NULL));
504:   }
505:   if (batchSize) *batchSize = Bbn;
506:   if (numBatches) *numBatches = numBb;
507:   PetscFunctionReturn(PETSC_SUCCESS);
508: }

510: static PetscErrorCode MatMatMultSymbolic_MPIAIJ_MPIDense(Mat A, Mat B, PetscReal fill, Mat C)
511: {
512:   Mat_MPIAIJ         *aij = (Mat_MPIAIJ *)A->data;
513:   MPIAIJ_MPIDense_AB *data;
514:   PetscInt            nz  = aij->B->cmap->n;
515:   VecScatter          ctx = aij->Mvctx;
516:   PetscInt            Am = A->rmap->n, BN = B->cmap->N, Bbn, numBb;
517:   Mat                 workB1, workC1;
518:   const char         *ctype;
519:   PetscBool           cisdense;

521:   PetscFunctionBegin;
522:   MatCheckProduct(C, 4);
523:   PetscCheck(!C->product->data, PetscObjectComm((PetscObject)C), PETSC_ERR_PLIB, "Product data not empty");
524:   PetscCall(PetscObjectBaseTypeCompare((PetscObject)C, MATMPIDENSE, &cisdense));
525:   if (!cisdense) {
526:     PetscCall(MatSetType(C, ((PetscObject)B)->type_name));
527:     PetscCall(MatSetVecType(C, B->defaultvectype));
528:   }
529:   PetscCall(MatSetSizes(C, Am, B->cmap->n, A->rmap->N, BN));
530:   PetscCall(MatSetBlockSizesFromMats(C, A, B));
531:   PetscCall(MatSetUp(C));
532:   /* The cuSPARSE and hipSPARSE symbolic below re-types a host local block of C in place, so capture the type
533:      of the local block of C now to give the work matrix the type the numeric will add it to */
534:   ctype = ((PetscObject)((Mat_MPIDense *)C->data)->A)->type_name;
535:   PetscCall(PetscNew(&data));
536:   PetscCall(MatMPIDenseScatterSetUp_Private(ctx, nz, 1, Am, B, C, &data->scatter, &Bbn, &numBb));
537:   PetscCall(MatSetOption(C, MAT_NO_OFF_PROC_ENTRIES, PETSC_TRUE));
538:   PetscCall(MatAssemblyBegin(C, MAT_FINAL_ASSEMBLY));
539:   PetscCall(MatAssemblyEnd(C, MAT_FINAL_ASSEMBLY));
540:   PetscCall(MatProductClear(aij->A));
541:   PetscCall(MatProductClear(((Mat_MPIDense *)B->data)->A));
542:   PetscCall(MatProductClear(((Mat_MPIDense *)C->data)->A));
543:   if (data->scatter.ondevice && nz) {
544:     /* On the device route the off-diagonal contribution is computed into a work matrix and added to the local
545:        block of C, rather than accumulated in place by a host kernel */
546:     PetscCall(MatCreate(PETSC_COMM_SELF, &data->workC));
547:     PetscCall(MatSetSizes(data->workC, Am, Bbn ? Bbn : BN, Am, Bbn ? Bbn : BN));
548:     PetscCall(MatSetType(data->workC, ctype));
549:     PetscCall(MatSetUp(data->workC));
550:     PetscCall(MatProductCreateWithMat(aij->B, data->scatter.workB, NULL, data->workC));
551:     PetscCall(MatProductSetType(data->workC, MATPRODUCT_AB));
552:     PetscCall(MatProductSetFromOptions(data->workC));
553:     PetscCall(MatProductSymbolic(data->workC));
554:     if (numBb && BN % Bbn) {
555:       /* the last column batch is smaller, set up the product on the sub-matrices the numeric will use for it */
556:       PetscCall(MatDenseGetSubMatrix(data->scatter.workB, PETSC_DECIDE, PETSC_DECIDE, 0, BN % Bbn, &workB1));
557:       PetscCall(MatDenseGetSubMatrix(data->workC, PETSC_DECIDE, PETSC_DECIDE, 0, BN % Bbn, &workC1));
558:       PetscCall(MatProductCreateWithMat(aij->B, workB1, NULL, workC1));
559:       PetscCall(MatProductSetType(workC1, MATPRODUCT_AB));
560:       PetscCall(MatProductSetFromOptions(workC1));
561:       PetscCall(MatProductSymbolic(workC1));
562:       PetscCall(MatDenseRestoreSubMatrix(data->workC, &workC1));
563:       PetscCall(MatDenseRestoreSubMatrix(data->scatter.workB, &workB1));
564:     }
565:   }
566:   PetscCall(MatProductCreateWithMat(aij->A, ((Mat_MPIDense *)B->data)->A, NULL, ((Mat_MPIDense *)C->data)->A));
567:   PetscCall(MatProductSetType(((Mat_MPIDense *)C->data)->A, MATPRODUCT_AB));
568:   PetscCall(MatProductSetFromOptions(((Mat_MPIDense *)C->data)->A));
569:   PetscCall(MatProductSymbolic(((Mat_MPIDense *)C->data)->A));
570:   C->product->data       = data;
571:   C->product->destroy    = MatMPIAIJ_MPIDenseDestroy;
572:   C->ops->matmultnumeric = MatMatMultNumeric_MPIAIJ_MPIDense;
573:   PetscFunctionReturn(PETSC_SUCCESS);
574: }

576: PETSC_INTERN PetscErrorCode MatMatMultNumericAdd_SeqAIJ_SeqDense(Mat, Mat, Mat, const PetscBool);

578: /*
579:     Performs an efficient scatter on the rows of B needed by this process; this is
580:     a modification of the VecScatterBegin_() routines.
581: */
582: static PetscErrorCode MatMPIDenseScatterHost_Private(VecScatter ctx, PetscInt bs, Mat workB, MPIAIJ_MPIDense *contents, Mat B, Mat C)
583: {
584:   const PetscScalar *b;
585:   PetscScalar       *rvalues;
586:   const PetscInt    *sindices, *sstarts, *rstarts;
587:   const PetscMPIInt *sprocs, *rprocs;
588:   PetscMPIInt        nsends, nrecvs;
589:   MPI_Comm           comm;
590:   PetscMPIInt        tag = ((PetscObject)ctx)->tag, ncols, nsends_mpi, nrecvs_mpi;

592:   PetscFunctionBegin;
593:   PetscCall(PetscMPIIntCast(B->cmap->N, &ncols));
594:   PetscCall(VecScatterGetRemote_Private(ctx, PETSC_TRUE /*send*/, &nsends, &sstarts, &sindices, &sprocs, NULL /*bs*/));
595:   PetscCall(VecScatterGetRemoteOrdered_Private(ctx, PETSC_FALSE /*recv*/, &nrecvs, &rstarts, NULL, &rprocs, NULL /*bs*/));
596:   PetscCall(PetscMPIIntCast(nsends, &nsends_mpi));
597:   PetscCall(PetscMPIIntCast(nrecvs, &nrecvs_mpi));

599:   PetscCall(MatDenseGetArrayRead(B, &b));
600:   PetscCall(MatDenseGetArray(workB, &rvalues));

602:   /* Post recv, use MPI derived data type to save memory */
603:   PetscCall(PetscObjectGetComm((PetscObject)C, &comm));
604:   for (PetscMPIInt i = 0; i < nrecvs; i++) PetscCallMPI(MPIU_Irecv(rvalues + ((rstarts[i] - rstarts[0]) * bs), ncols, contents->rtype[i], rprocs[i], tag, comm, contents->rwaits + i));
605:   for (PetscMPIInt i = 0; i < nsends; i++) PetscCallMPI(MPIU_Isend(b, ncols, contents->stype[i], sprocs[i], tag, comm, contents->swaits + i));

607:   if (nrecvs) PetscCallMPI(MPI_Waitall(nrecvs_mpi, contents->rwaits, MPI_STATUSES_IGNORE));
608:   if (nsends) PetscCallMPI(MPI_Waitall(nsends_mpi, contents->swaits, MPI_STATUSES_IGNORE));

610:   PetscCall(VecScatterRestoreRemote_Private(ctx, PETSC_TRUE /*send*/, &nsends, &sstarts, &sindices, &sprocs, NULL));
611:   PetscCall(VecScatterRestoreRemoteOrdered_Private(ctx, PETSC_FALSE /*recv*/, &nrecvs, &rstarts, NULL, &rprocs, NULL));
612:   PetscCall(MatDenseRestoreArrayRead(B, &b));
613:   PetscCall(MatDenseRestoreArray(workB, &rvalues));
614:   PetscFunctionReturn(PETSC_SUCCESS);
615: }

617: /*
618:    Same as MatMPIDenseScatterHost_Private(), but using the strided PetscSFs set up in
619:    MatMPIDenseScatterSetUp_Private() so that the data never leaves the device. The scatter is split in two
620:    halves so that a caller can run other work while the off-process rows of B are in flight
621: */
622: static PetscErrorCode MatMPIDenseScatterDeviceBegin_Private(Mat workB, MPIAIJ_MPIDense *contents, Mat B)
623: {
624:   PetscSF      sf;
625:   PetscInt     k = 0;
626:   PetscMemType bmtype, wmtype;

628:   PetscFunctionBegin;
629:   PetscCheck(!contents->sfinuse, PETSC_COMM_SELF, PETSC_ERR_PLIB, "A scatter is already in flight");
630:   if (B->cmap->N != contents->ncols[0]) k = 1;
631:   PetscCheck(contents->sf[k] && contents->ncols[k] == B->cmap->N, PETSC_COMM_SELF, PETSC_ERR_PLIB, "No scatter set up for %" PetscInt_FMT " columns", B->cmap->N);
632:   sf = contents->sf[k];
633:   /* every entry of workB is overwritten, so write-only access is enough */
634:   PetscCall(MatDenseGetArrayReadAndMemType(B, &contents->barray, &bmtype));
635:   PetscCall(MatDenseGetArrayWriteAndMemType(workB, &contents->warray, &wmtype));
636:   PetscCall(PetscSFBcastWithMemTypeBegin(sf, sf->vscat.unit, bmtype, contents->barray, wmtype, contents->warray, MPI_REPLACE));
637:   contents->sfinuse = sf;
638:   PetscFunctionReturn(PETSC_SUCCESS);
639: }

641: static PetscErrorCode MatMPIDenseScatterDeviceEnd_Private(Mat workB, MPIAIJ_MPIDense *contents, Mat B)
642: {
643:   PetscSF sf = contents->sfinuse;

645:   PetscFunctionBegin;
646:   PetscCheck(sf, PETSC_COMM_SELF, PETSC_ERR_PLIB, "No scatter in flight");
647:   PetscCall(PetscSFBcastEnd(sf, sf->vscat.unit, contents->barray, contents->warray, MPI_REPLACE));
648:   PetscCall(MatDenseRestoreArrayWriteAndMemType(workB, &contents->warray));
649:   PetscCall(MatDenseRestoreArrayReadAndMemType(B, &contents->barray));
650:   contents->sfinuse = NULL;
651:   PetscFunctionReturn(PETSC_SUCCESS);
652: }

654: /*
655:    The checks the host and device scatters share; on the device route they are done by the begin half
656: */
657: static PetscErrorCode MatMPIDenseScatterCheck_Private(PetscInt nrows, Mat workB, MPIAIJ_MPIDense *contents, Mat B, Mat C)
658: {
659:   PetscInt blda;

661:   PetscFunctionBegin;
662:   MatCheckProduct(C, 5);
663:   PetscCheck(C->product->data, PetscObjectComm((PetscObject)C), PETSC_ERR_PLIB, "Product data empty");
664:   PetscCheck(nrows == workB->rmap->n, PETSC_COMM_SELF, PETSC_ERR_PLIB, "Number of rows of workB %" PetscInt_FMT " not equal to columns of off-diagonal block %" PetscInt_FMT, workB->rmap->n, nrows);
665:   PetscCall(MatDenseGetLDA(B, &blda));
666:   PetscCheck(blda == contents->blda, PETSC_COMM_SELF, PETSC_ERR_ARG_WRONG, "Cannot reuse an input matrix with lda %" PetscInt_FMT " != %" PetscInt_FMT, blda, contents->blda);
667:   PetscFunctionReturn(PETSC_SUCCESS);
668: }

670: PETSC_INTERN PetscErrorCode MatMPIDenseScatter_Private(VecScatter ctx, PetscInt nrows, PetscInt bs, Mat workB, MPIAIJ_MPIDense *contents, Mat B, Mat C)
671: {
672:   PetscFunctionBegin;
673:   PetscCall(MatMPIDenseScatterCheck_Private(nrows, workB, contents, B, C));
674:   if (contents->ondevice) {
675:     PetscCall(MatMPIDenseScatterDeviceBegin_Private(workB, contents, B));
676:     PetscCall(MatMPIDenseScatterDeviceEnd_Private(workB, contents, B));
677:   } else PetscCall(MatMPIDenseScatterHost_Private(ctx, bs, workB, contents, B, C));
678:   PetscFunctionReturn(PETSC_SUCCESS);
679: }

681: /*
682:    Starts the scatter of the off-process rows of B into workB. On the device route only the begin half of the
683:    PetscSF broadcast is posted, so the caller must complete it with MatMPIDenseScatterEnd(); the host scatter
684:    is blocking and is done entirely here
685: */
686: static PetscErrorCode MatMPIDenseScatterBegin(Mat A, Mat B, Mat workB, Mat C)
687: {
688:   Mat_MPIAIJ      *aij      = (Mat_MPIAIJ *)A->data;
689:   MPIAIJ_MPIDense *contents = &((MPIAIJ_MPIDense_AB *)C->product->data)->scatter;

691:   PetscFunctionBegin;
692:   if (contents->ondevice) {
693:     PetscCall(MatMPIDenseScatterCheck_Private(aij->B->cmap->n, workB, contents, B, C));
694:     PetscCall(MatMPIDenseScatterDeviceBegin_Private(workB, contents, B));
695:   } else PetscCall(MatMPIDenseScatter_Private(aij->Mvctx, aij->B->cmap->n, 1, workB, contents, B, C));
696:   PetscFunctionReturn(PETSC_SUCCESS);
697: }

699: static PetscErrorCode MatMPIDenseScatterEnd(Mat B, Mat workB, Mat C)
700: {
701:   MPIAIJ_MPIDense *contents = &((MPIAIJ_MPIDense_AB *)C->product->data)->scatter;

703:   PetscFunctionBegin;
704:   if (contents->ondevice) PetscCall(MatMPIDenseScatterDeviceEnd_Private(workB, contents, B));
705:   PetscFunctionReturn(PETSC_SUCCESS);
706: }

708: /*
709:    Computes C = A * B, creating the nested product and its symbolic the first time. When clear is true the
710:    product is cleared by the numeric, so that the symbolic is redone on the next call
711: */
712: static PetscErrorCode MatMPIAIJ_MPIDenseProductNumeric_Private(Mat A, Mat B, Mat C, PetscBool clear)
713: {
714:   PetscFunctionBegin;
715:   if (!C->product) {
716:     PetscCall(MatProductCreateWithMat(A, B, NULL, C));
717:     PetscCall(MatProductSetType(C, MATPRODUCT_AB));
718:     PetscCall(MatProductSetFromOptions(C));
719:     PetscCall(MatProductSymbolic(C));
720:   } else PetscCall(MatProductReplaceMats(A, B, NULL, C));
721:   if (clear && !C->product->clear) C->product->clear = PETSC_TRUE;
722:   PetscCall(MatProductNumeric(C));
723:   PetscFunctionReturn(PETSC_SUCCESS);
724: }

726: static PetscErrorCode MatMatMultNumeric_MPIAIJ_MPIDense(Mat A, Mat B, Mat C)
727: {
728:   Mat_MPIAIJ         *aij    = (Mat_MPIAIJ *)A->data;
729:   Mat_MPIDense       *bdense = (Mat_MPIDense *)B->data;
730:   Mat_MPIDense       *cdense = (Mat_MPIDense *)C->data;
731:   Mat                 workB;
732:   MPIAIJ_MPIDense    *contents;
733:   MPIAIJ_MPIDense_AB *data;
734:   PetscBool           clear = PETSC_FALSE, flg, overlap;

736:   PetscFunctionBegin;
737:   MatCheckProduct(C, 3);
738:   PetscCheck(C->product->data, PetscObjectComm((PetscObject)C), PETSC_ERR_PLIB, "Product data empty");
739:   data     = (MPIAIJ_MPIDense_AB *)C->product->data;
740:   contents = &data->scatter;
741:   if (PetscDefined(HAVE_CUPM)) {
742:     PetscCall(PetscObjectTypeCompare((PetscObject)C, MATMPIDENSE, &flg));
743:     if (flg) PetscCall(PetscObjectTypeCompare((PetscObject)A, MATMPIAIJ, &flg));
744:     clear = (PetscBool)!flg; /* if either A or C is a device Mat, make sure MatProductClear() is called */
745:   }
746:   /* The device scatter is non-blocking, so when there are no column batches start it before the product of the
747:      diagonal block below to overlap the communication with that product, as MatMult_MPIAIJ() does. The host
748:      scatter is blocking, so it is left where it is */
749:   overlap = (PetscBool)(contents->ondevice && contents->workB->cmap->n == B->cmap->N);
750:   if (overlap) PetscCall(MatMPIDenseScatterBegin(A, B, contents->workB, C));
751:   /* diagonal block of A times all local rows of B, first make sure that everything is up-to-date */
752:   PetscCall(MatMPIAIJ_MPIDenseProductNumeric_Private(aij->A, bdense->A, cdense->A, clear));
753:   if (contents->workB->cmap->n == B->cmap->N) {
754:     /* get off processor parts of B needed to complete C=A*B */
755:     workB = contents->workB;
756:     if (!overlap) PetscCall(MatMPIDenseScatterBegin(A, B, workB, C));
757:     PetscCall(MatMPIDenseScatterEnd(B, workB, C));

759:     /* off-diagonal block of A times nonlocal rows of B */
760:     if (data->workC) {
761:       PetscCall(MatMPIAIJ_MPIDenseProductNumeric_Private(aij->B, workB, data->workC, clear));
762:       PetscCall(MatAXPY(cdense->A, 1.0, data->workC, SAME_NONZERO_PATTERN));
763:     } else if (!contents->ondevice) PetscCall(MatMatMultNumericAdd_SeqAIJ_SeqDense(aij->B, workB, cdense->A, PETSC_TRUE)); /* the device route has no work matrix only when the off-diagonal block has no columns */
764:   } else {
765:     Mat       Bb, Cb, workC;
766:     PetscInt  BN = B->cmap->N, n = contents->workB->cmap->n, cols;
767:     PetscBool ccpu = PETSC_FALSE;

769:     PetscCheck(n > 0, PETSC_COMM_SELF, PETSC_ERR_ARG_WRONG, "Column block size %" PetscInt_FMT " must be positive", n);
770:     /* The host route accumulates into the local block of C on the host, bind C to the CPU to avoid copies back
771:        and forth from the device when getting and restoring the sub-matrices */
772:     if (!contents->ondevice) {
773:       PetscCall(MatBoundToCPU(C, &ccpu));
774:       PetscCall(MatBindToCPU(C, PETSC_TRUE));
775:     }
776:     for (PetscInt i = 0; i < BN; i += n) {
777:       cols  = PetscMin(n, BN - i);
778:       workB = contents->workB;
779:       workC = data->workC;
780:       if (cols != n) {
781:         PetscCall(MatDenseGetSubMatrix(contents->workB, PETSC_DECIDE, PETSC_DECIDE, 0, cols, &workB));
782:         if (workC) PetscCall(MatDenseGetSubMatrix(data->workC, PETSC_DECIDE, PETSC_DECIDE, 0, cols, &workC));
783:       }
784:       PetscCall(MatDenseGetSubMatrix(B, PETSC_DECIDE, PETSC_DECIDE, i, i + cols, &Bb));
785:       PetscCall(MatDenseGetSubMatrix(C, PETSC_DECIDE, PETSC_DECIDE, i, i + cols, &Cb));

787:       /* get off processor parts of B needed to complete C=A*B */
788:       PetscCall(MatMPIDenseScatterBegin(A, Bb, workB, C));
789:       PetscCall(MatMPIDenseScatterEnd(Bb, workB, C));

791:       /* off-diagonal block of A times nonlocal rows of B */
792:       if (workC) {
793:         PetscCall(MatMPIAIJ_MPIDenseProductNumeric_Private(aij->B, workB, workC, clear));
794:         PetscCall(MatAXPY(((Mat_MPIDense *)Cb->data)->A, 1.0, workC, SAME_NONZERO_PATTERN));
795:       } else if (!contents->ondevice) PetscCall(MatMatMultNumericAdd_SeqAIJ_SeqDense(aij->B, workB, ((Mat_MPIDense *)Cb->data)->A, PETSC_TRUE));
796:       if (cols != n) {
797:         if (workC) PetscCall(MatDenseRestoreSubMatrix(data->workC, &workC));
798:         PetscCall(MatDenseRestoreSubMatrix(contents->workB, &workB));
799:       }
800:       PetscCall(MatDenseRestoreSubMatrix(B, &Bb));
801:       PetscCall(MatDenseRestoreSubMatrix(C, &Cb));
802:     }
803:     if (!contents->ondevice) PetscCall(MatBindToCPU(C, ccpu));
804:   }
805:   PetscFunctionReturn(PETSC_SUCCESS);
806: }

808: PetscErrorCode MatMatMultNumeric_MPIAIJ_MPIAIJ(Mat A, Mat P, Mat C)
809: {
810:   Mat_MPIAIJ          *a = (Mat_MPIAIJ *)A->data, *c = (Mat_MPIAIJ *)C->data;
811:   Mat_SeqAIJ          *ad = (Mat_SeqAIJ *)a->A->data, *ao = (Mat_SeqAIJ *)a->B->data;
812:   Mat_SeqAIJ          *cd = (Mat_SeqAIJ *)c->A->data, *co = (Mat_SeqAIJ *)c->B->data;
813:   PetscInt            *adi = ad->i, *adj, *aoi = ao->i, *aoj;
814:   PetscScalar         *ada, *aoa, *cda = cd->a, *coa = co->a;
815:   Mat_SeqAIJ          *p_loc, *p_oth;
816:   PetscInt            *pi_loc, *pj_loc, *pi_oth, *pj_oth, *pj;
817:   PetscScalar         *pa_loc, *pa_oth, *pa, valtmp, *ca;
818:   PetscInt             cm = C->rmap->n, anz, pnz;
819:   MatProductCtx_APMPI *ptap;
820:   PetscScalar         *apa_sparse;
821:   const PetscScalar   *dummy;
822:   PetscInt            *api, *apj, *apJ, i, j, k, row;
823:   PetscInt             cstart = C->cmap->rstart;
824:   PetscInt             cdnz, conz, k0, k1, nextp;
825:   MPI_Comm             comm;
826:   PetscMPIInt          size;

828:   PetscFunctionBegin;
829:   MatCheckProduct(C, 3);
830:   ptap = (MatProductCtx_APMPI *)C->product->data;
831:   PetscCheck(ptap, PetscObjectComm((PetscObject)C), PETSC_ERR_ARG_WRONGSTATE, "PtAP cannot be computed. Missing data");
832:   PetscCall(PetscObjectGetComm((PetscObject)C, &comm));
833:   PetscCallMPI(MPI_Comm_size(comm, &size));
834:   PetscCheck(ptap->P_oth || size <= 1, PetscObjectComm((PetscObject)C), PETSC_ERR_ARG_WRONGSTATE, "AP cannot be reused. Do not call MatProductClear()");

836:   /* flag CPU mask for C */
837: #if PetscDefined(HAVE_DEVICE)
838:   if (C->offloadmask != PETSC_OFFLOAD_UNALLOCATED) C->offloadmask = PETSC_OFFLOAD_CPU;
839:   if (c->A->offloadmask != PETSC_OFFLOAD_UNALLOCATED) c->A->offloadmask = PETSC_OFFLOAD_CPU;
840:   if (c->B->offloadmask != PETSC_OFFLOAD_UNALLOCATED) c->B->offloadmask = PETSC_OFFLOAD_CPU;
841: #endif
842:   apa_sparse = ptap->apa;

844:   /* 1) get P_oth = ptap->P_oth  and P_loc = ptap->P_loc */
845:   /* update numerical values of P_oth and P_loc */
846:   PetscCall(MatGetBrowsOfAoCols_MPIAIJ(A, P, MAT_REUSE_MATRIX, &ptap->startsj_s, &ptap->startsj_r, &ptap->bufa, &ptap->P_oth));
847:   PetscCall(MatMPIAIJGetLocalMat(P, MAT_REUSE_MATRIX, &ptap->P_loc));

849:   /* 2) compute numeric C_loc = A_loc*P = Ad*P_loc + Ao*P_oth */
850:   /* get data from symbolic products */
851:   p_loc  = (Mat_SeqAIJ *)ptap->P_loc->data;
852:   pi_loc = p_loc->i;
853:   pj_loc = p_loc->j;
854:   pa_loc = p_loc->a;
855:   if (size > 1) {
856:     p_oth  = (Mat_SeqAIJ *)ptap->P_oth->data;
857:     pi_oth = p_oth->i;
858:     pj_oth = p_oth->j;
859:     pa_oth = p_oth->a;
860:   } else {
861:     p_oth  = NULL;
862:     pi_oth = NULL;
863:     pj_oth = NULL;
864:     pa_oth = NULL;
865:   }

867:   /* trigger copy to CPU */
868:   PetscCall(MatSeqAIJGetArrayRead(a->A, &dummy));
869:   PetscCall(MatSeqAIJRestoreArrayRead(a->A, &dummy));
870:   PetscCall(MatSeqAIJGetArrayRead(a->B, &dummy));
871:   PetscCall(MatSeqAIJRestoreArrayRead(a->B, &dummy));
872:   api = ptap->api;
873:   apj = ptap->apj;
874:   for (i = 0; i < cm; i++) {
875:     apJ = apj + api[i];

877:     /* diagonal portion of A */
878:     anz = adi[i + 1] - adi[i];
879:     adj = ad->j + adi[i];
880:     ada = ad->a + adi[i];
881:     for (j = 0; j < anz; j++) {
882:       row = adj[j];
883:       pnz = pi_loc[row + 1] - pi_loc[row];
884:       pj  = pj_loc + pi_loc[row];
885:       pa  = pa_loc + pi_loc[row];
886:       /* perform sparse axpy */
887:       valtmp = ada[j];
888:       nextp  = 0;
889:       for (k = 0; nextp < pnz; k++) {
890:         if (apJ[k] == pj[nextp]) { /* column of AP == column of P */
891:           apa_sparse[k] += valtmp * pa[nextp++];
892:         }
893:       }
894:       PetscCall(PetscLogFlops(2.0 * pnz));
895:     }

897:     /* off-diagonal portion of A */
898:     anz = aoi[i + 1] - aoi[i];
899:     aoj = PetscSafePointerPlusOffset(ao->j, aoi[i]);
900:     aoa = PetscSafePointerPlusOffset(ao->a, aoi[i]);
901:     for (j = 0; j < anz; j++) {
902:       row = aoj[j];
903:       pnz = pi_oth[row + 1] - pi_oth[row];
904:       pj  = pj_oth + pi_oth[row];
905:       pa  = pa_oth + pi_oth[row];
906:       /* perform sparse axpy */
907:       valtmp = aoa[j];
908:       nextp  = 0;
909:       for (k = 0; nextp < pnz; k++) {
910:         if (apJ[k] == pj[nextp]) { /* column of AP == column of P */
911:           apa_sparse[k] += valtmp * pa[nextp++];
912:         }
913:       }
914:       PetscCall(PetscLogFlops(2.0 * pnz));
915:     }

917:     /* set values in C */
918:     cdnz = cd->i[i + 1] - cd->i[i];
919:     conz = co->i[i + 1] - co->i[i];

921:     /* 1st off-diagonal part of C */
922:     ca = PetscSafePointerPlusOffset(coa, co->i[i]);
923:     k  = 0;
924:     for (k0 = 0; k0 < conz; k0++) {
925:       if (apJ[k] >= cstart) break;
926:       ca[k0]        = apa_sparse[k];
927:       apa_sparse[k] = 0.0;
928:       k++;
929:     }

931:     /* diagonal part of C */
932:     ca = cda + cd->i[i];
933:     for (k1 = 0; k1 < cdnz; k1++) {
934:       ca[k1]        = apa_sparse[k];
935:       apa_sparse[k] = 0.0;
936:       k++;
937:     }

939:     /* 2nd off-diagonal part of C */
940:     ca = PetscSafePointerPlusOffset(coa, co->i[i]);
941:     for (; k0 < conz; k0++) {
942:       ca[k0]        = apa_sparse[k];
943:       apa_sparse[k] = 0.0;
944:       k++;
945:     }
946:   }
947:   PetscCall(MatAssemblyBegin(C, MAT_FINAL_ASSEMBLY));
948:   PetscCall(MatAssemblyEnd(C, MAT_FINAL_ASSEMBLY));
949:   PetscFunctionReturn(PETSC_SUCCESS);
950: }

952: /* same as MatMatMultSymbolic_MPIAIJ_MPIAIJ_nonscalable(), except using LLCondensed to avoid O(BN) memory requirement */
953: PetscErrorCode MatMatMultSymbolic_MPIAIJ_MPIAIJ(Mat A, Mat P, PetscReal fill, Mat C)
954: {
955:   MPI_Comm             comm;
956:   PetscMPIInt          size;
957:   MatProductCtx_APMPI *ptap;
958:   PetscFreeSpaceList   free_space = NULL, current_space = NULL;
959:   Mat_MPIAIJ          *a  = (Mat_MPIAIJ *)A->data;
960:   Mat_SeqAIJ          *ad = (Mat_SeqAIJ *)a->A->data, *ao = (Mat_SeqAIJ *)a->B->data, *p_loc, *p_oth;
961:   PetscInt            *pi_loc, *pj_loc, *pi_oth, *pj_oth, *dnz, *onz;
962:   PetscInt            *adi = ad->i, *adj = ad->j, *aoi = ao->i, *aoj = ao->j, rstart = A->rmap->rstart;
963:   PetscInt             i, pnz, row, *api, *apj, *Jptr, apnz, nspacedouble = 0, j, nzi, *lnk, apnz_max = 1;
964:   PetscInt             am = A->rmap->n, pn = P->cmap->n, pm = P->rmap->n, lsize = pn + 20;
965:   PetscReal            afill;
966:   MatType              mtype;

968:   PetscFunctionBegin;
969:   MatCheckProduct(C, 4);
970:   PetscCheck(!C->product->data, PETSC_COMM_SELF, PETSC_ERR_PLIB, "Extra product struct not empty");
971:   PetscCall(PetscObjectGetComm((PetscObject)A, &comm));
972:   PetscCallMPI(MPI_Comm_size(comm, &size));

974:   /* create struct MatProductCtx_APMPI and attached it to C later */
975:   PetscCall(PetscNew(&ptap));

977:   /* get P_oth by taking rows of P (= non-zero cols of local A) from other processors */
978:   PetscCall(MatGetBrowsOfAoCols_MPIAIJ_Private(A, P, MAT_INITIAL_MATRIX, C->structure_only, &ptap->startsj_s, &ptap->startsj_r, &ptap->bufa, &ptap->P_oth));

980:   /* get P_loc by taking all local rows of P */
981:   PetscCall(MatMPIAIJGetLocalMat_Private(P, MAT_INITIAL_MATRIX, C->structure_only, &ptap->P_loc));

983:   p_loc  = (Mat_SeqAIJ *)ptap->P_loc->data;
984:   pi_loc = p_loc->i;
985:   pj_loc = p_loc->j;
986:   if (size > 1) {
987:     p_oth  = (Mat_SeqAIJ *)ptap->P_oth->data;
988:     pi_oth = p_oth->i;
989:     pj_oth = p_oth->j;
990:   } else {
991:     p_oth  = NULL;
992:     pi_oth = NULL;
993:     pj_oth = NULL;
994:   }

996:   /* first, compute symbolic AP = A_loc*P = A_diag*P_loc + A_off*P_oth */
997:   PetscCall(PetscMalloc1(am + 1, &api));
998:   ptap->api = api;
999:   api[0]    = 0;

1001:   PetscCall(PetscLLCondensedCreate_Scalable(lsize, &lnk));

1003:   /* Initial FreeSpace size is fill*(nnz(A)+nnz(P)) */
1004:   PetscCall(PetscFreeSpaceGet(PetscRealIntMultTruncate(fill, PetscIntSumTruncate(adi[am], PetscIntSumTruncate(aoi[am], pi_loc[pm]))), &free_space));
1005:   current_space = free_space;
1006:   MatPreallocateBegin(comm, am, pn, dnz, onz);
1007:   for (i = 0; i < am; i++) {
1008:     /* diagonal portion of A */
1009:     nzi = adi[i + 1] - adi[i];
1010:     for (j = 0; j < nzi; j++) {
1011:       row  = *adj++;
1012:       pnz  = pi_loc[row + 1] - pi_loc[row];
1013:       Jptr = pj_loc + pi_loc[row];
1014:       /* Expand list if it is not long enough */
1015:       if (pnz + apnz_max > lsize) {
1016:         lsize = pnz + apnz_max;
1017:         PetscCall(PetscLLCondensedExpand_Scalable(lsize, &lnk));
1018:       }
1019:       /* add non-zero cols of P into the sorted linked list lnk */
1020:       PetscCall(PetscLLCondensedAddSorted_Scalable(pnz, Jptr, lnk));
1021:       apnz       = *lnk; /* The first element in the list is the number of items in the list */
1022:       api[i + 1] = api[i] + apnz;
1023:       if (apnz > apnz_max) apnz_max = apnz + 1; /* '1' for diagonal entry */
1024:     }
1025:     /* off-diagonal portion of A */
1026:     nzi = aoi[i + 1] - aoi[i];
1027:     for (j = 0; j < nzi; j++) {
1028:       row  = *aoj++;
1029:       pnz  = pi_oth[row + 1] - pi_oth[row];
1030:       Jptr = pj_oth + pi_oth[row];
1031:       /* Expand list if it is not long enough */
1032:       if (pnz + apnz_max > lsize) {
1033:         lsize = pnz + apnz_max;
1034:         PetscCall(PetscLLCondensedExpand_Scalable(lsize, &lnk));
1035:       }
1036:       /* add non-zero cols of P into the sorted linked list lnk */
1037:       PetscCall(PetscLLCondensedAddSorted_Scalable(pnz, Jptr, lnk));
1038:       apnz       = *lnk; /* The first element in the list is the number of items in the list */
1039:       api[i + 1] = api[i] + apnz;
1040:       if (apnz > apnz_max) apnz_max = apnz + 1; /* '1' for diagonal entry */
1041:     }

1043:     /* add missing diagonal entry */
1044:     if (C->force_diagonals) {
1045:       j = i + rstart; /* column index */
1046:       PetscCall(PetscLLCondensedAddSorted_Scalable(1, &j, lnk));
1047:     }

1049:     apnz       = *lnk;
1050:     api[i + 1] = api[i] + apnz;
1051:     if (apnz > apnz_max) apnz_max = apnz;

1053:     /* if free space is not available, double the total space in the list */
1054:     if (current_space->local_remaining < apnz) {
1055:       PetscCall(PetscFreeSpaceGet(PetscIntSumTruncate(apnz, current_space->total_array_size), &current_space));
1056:       nspacedouble++;
1057:     }

1059:     /* Copy data into free space, then initialize lnk */
1060:     PetscCall(PetscLLCondensedClean_Scalable(apnz, current_space->array, lnk));
1061:     PetscCall(MatPreallocateSet(i + rstart, apnz, current_space->array, dnz, onz));

1063:     current_space->array += apnz;
1064:     current_space->local_used += apnz;
1065:     current_space->local_remaining -= apnz;
1066:   }

1068:   /* Allocate space for apj, initialize apj, and */
1069:   /* destroy list of free space and other temporary array(s) */
1070:   PetscCall(PetscMalloc1(api[am], &ptap->apj));
1071:   apj = ptap->apj;
1072:   PetscCall(PetscFreeSpaceContiguous(&free_space, ptap->apj));
1073:   PetscCall(PetscLLCondensedDestroy_Scalable(lnk));

1075:   /* create and assemble symbolic parallel matrix C */
1076:   PetscCall(MatSetSizes(C, am, pn, PETSC_DETERMINE, PETSC_DETERMINE));
1077:   PetscCall(MatSetBlockSizesFromMats(C, A, P));
1078:   PetscCall(MatGetType(A, &mtype));
1079:   PetscCall(MatSetType(C, C->structure_only ? MATMPIAIJ : mtype));
1080:   PetscCall(MatMPIAIJSetPreallocation(C, 0, dnz, 0, onz));
1081:   MatPreallocateEnd(dnz, onz);

1083:   /* malloc apa for assembly C */
1084:   if (!C->structure_only) PetscCall(PetscCalloc1(apnz_max, &ptap->apa));

1086:   PetscCall(MatSetValues_MPIAIJ_CopyFromCSRFormat_Symbolic(C, apj, api));
1087:   PetscCall(MatSetOption(C, MAT_NO_OFF_PROC_ENTRIES, PETSC_TRUE));
1088:   PetscCall(MatAssemblyBegin(C, MAT_FINAL_ASSEMBLY));
1089:   PetscCall(MatAssemblyEnd(C, MAT_FINAL_ASSEMBLY));
1090:   PetscCall(MatSetOption(C, MAT_NEW_NONZERO_LOCATION_ERR, PETSC_TRUE));

1092:   C->ops->matmultnumeric = MatMatMultNumeric_MPIAIJ_MPIAIJ;
1093:   C->ops->productnumeric = MatProductNumeric_AB;

1095:   /* attach the supporting struct to C for reuse */
1096:   C->product->data    = ptap;
1097:   C->product->destroy = MatProductCtxDestroy_MPIAIJ_MatMatMult;

1099:   /* set MatInfo */
1100:   afill = (PetscReal)api[am] / (adi[am] + aoi[am] + pi_loc[pm] + 1) + 1.e-5;
1101:   if (afill < 1.0) afill = 1.0;
1102:   C->info.mallocs           = nspacedouble;
1103:   C->info.fill_ratio_given  = fill;
1104:   C->info.fill_ratio_needed = afill;

1106:   if (PetscDefined(USE_INFO)) {
1107:     if (api[am]) {
1108:       PetscCall(PetscInfo(C, "Reallocs %" PetscInt_FMT "; Fill ratio: given %g needed %g.\n", nspacedouble, (double)fill, (double)afill));
1109:       PetscCall(PetscInfo(C, "Use MatMatMult(A,B,MatReuse,%g,&C) for best performance.;\n", (double)afill));
1110:     } else PetscCall(PetscInfo(C, "Empty matrix product\n"));
1111:   }
1112:   PetscFunctionReturn(PETSC_SUCCESS);
1113: }

1115: /* This function is needed for the seqMPI matrix-matrix multiplication.  */
1116: /* Three input arrays are merged to one output array. The size of the    */
1117: /* output array is also output. Duplicate entries only show up once.     */
1118: static void Merge3SortedArrays(PetscInt size1, PetscInt *in1, PetscInt size2, PetscInt *in2, PetscInt size3, PetscInt *in3, PetscInt *size4, PetscInt *out)
1119: {
1120:   int i = 0, j = 0, k = 0, l = 0;

1122:   /* Traverse all three arrays */
1123:   while (i < size1 && j < size2 && k < size3) {
1124:     if (in1[i] < in2[j] && in1[i] < in3[k]) {
1125:       out[l++] = in1[i++];
1126:     } else if (in2[j] < in1[i] && in2[j] < in3[k]) {
1127:       out[l++] = in2[j++];
1128:     } else if (in3[k] < in1[i] && in3[k] < in2[j]) {
1129:       out[l++] = in3[k++];
1130:     } else if (in1[i] == in2[j] && in1[i] < in3[k]) {
1131:       out[l++] = in1[i];
1132:       i++, j++;
1133:     } else if (in1[i] == in3[k] && in1[i] < in2[j]) {
1134:       out[l++] = in1[i];
1135:       i++, k++;
1136:     } else if (in3[k] == in2[j] && in2[j] < in1[i]) {
1137:       out[l++] = in2[j];
1138:       k++, j++;
1139:     } else if (in1[i] == in2[j] && in1[i] == in3[k]) {
1140:       out[l++] = in1[i];
1141:       i++, j++, k++;
1142:     }
1143:   }

1145:   /* Traverse two remaining arrays */
1146:   while (i < size1 && j < size2) {
1147:     if (in1[i] < in2[j]) {
1148:       out[l++] = in1[i++];
1149:     } else if (in1[i] > in2[j]) {
1150:       out[l++] = in2[j++];
1151:     } else {
1152:       out[l++] = in1[i];
1153:       i++, j++;
1154:     }
1155:   }

1157:   while (i < size1 && k < size3) {
1158:     if (in1[i] < in3[k]) {
1159:       out[l++] = in1[i++];
1160:     } else if (in1[i] > in3[k]) {
1161:       out[l++] = in3[k++];
1162:     } else {
1163:       out[l++] = in1[i];
1164:       i++, k++;
1165:     }
1166:   }

1168:   while (k < size3 && j < size2) {
1169:     if (in3[k] < in2[j]) {
1170:       out[l++] = in3[k++];
1171:     } else if (in3[k] > in2[j]) {
1172:       out[l++] = in2[j++];
1173:     } else {
1174:       out[l++] = in3[k];
1175:       k++, j++;
1176:     }
1177:   }

1179:   /* Traverse one remaining array */
1180:   while (i < size1) out[l++] = in1[i++];
1181:   while (j < size2) out[l++] = in2[j++];
1182:   while (k < size3) out[l++] = in3[k++];

1184:   *size4 = l;
1185: }

1187: /* This matrix-matrix multiplication algorithm divides the multiplication into three multiplications and  */
1188: /* adds up the products. Two of these three multiplications are performed with existing (sequential)      */
1189: /* matrix-matrix multiplications.  */
1190: PetscErrorCode MatMatMultSymbolic_MPIAIJ_MPIAIJ_seqMPI(Mat A, Mat P, PetscReal fill, Mat C)
1191: {
1192:   MPI_Comm             comm;
1193:   PetscMPIInt          size;
1194:   MatProductCtx_APMPI *ptap;
1195:   PetscFreeSpaceList   free_space_diag = NULL, current_space = NULL;
1196:   Mat_MPIAIJ          *a  = (Mat_MPIAIJ *)A->data;
1197:   Mat_SeqAIJ          *ad = (Mat_SeqAIJ *)a->A->data, *ao = (Mat_SeqAIJ *)a->B->data, *p_loc;
1198:   Mat_MPIAIJ          *p = (Mat_MPIAIJ *)P->data;
1199:   Mat_SeqAIJ          *adpd_seq, *p_off, *aopoth_seq;
1200:   PetscInt             adponz, adpdnz;
1201:   PetscInt            *pi_loc, *dnz, *onz;
1202:   PetscInt            *adi = ad->i, *adj = ad->j, *aoi = ao->i, rstart = A->rmap->rstart;
1203:   PetscInt            *lnk, i, i1 = 0, pnz, row, *adpoi, *adpoj, *api, *adpoJ, *aopJ, *apJ, *Jptr, aopnz, nspacedouble = 0, j, nzi, *apj, apnz, *adpdi, *adpdj, *adpdJ, *poff_i, *poff_j, *j_temp, *aopothi, *aopothj;
1204:   PetscInt             am = A->rmap->n, pN = P->cmap->N, pn = P->cmap->n, pm = P->rmap->n, p_colstart, p_colend;
1205:   PetscBT              lnkbt;
1206:   PetscReal            afill;
1207:   PetscMPIInt          rank;
1208:   Mat                  adpd, aopoth;
1209:   MatType              mtype;
1210:   const char          *prefix;

1212:   PetscFunctionBegin;
1213:   MatCheckProduct(C, 4);
1214:   PetscCheck(!C->product->data, PETSC_COMM_SELF, PETSC_ERR_PLIB, "Extra product struct not empty");
1215:   PetscCall(PetscObjectGetComm((PetscObject)A, &comm));
1216:   PetscCallMPI(MPI_Comm_size(comm, &size));
1217:   PetscCallMPI(MPI_Comm_rank(comm, &rank));
1218:   PetscCall(MatGetOwnershipRangeColumn(P, &p_colstart, &p_colend));

1220:   /* create struct MatProductCtx_APMPI and attached it to C later */
1221:   PetscCall(PetscNew(&ptap));

1223:   /* get P_oth by taking rows of P (= non-zero cols of local A) from other processors */
1224:   PetscCall(MatGetBrowsOfAoCols_MPIAIJ(A, P, MAT_INITIAL_MATRIX, &ptap->startsj_s, &ptap->startsj_r, &ptap->bufa, &ptap->P_oth));

1226:   /* get P_loc by taking all local rows of P */
1227:   PetscCall(MatMPIAIJGetLocalMat(P, MAT_INITIAL_MATRIX, &ptap->P_loc));

1229:   p_loc  = (Mat_SeqAIJ *)ptap->P_loc->data;
1230:   pi_loc = p_loc->i;

1232:   /* Allocate memory for the i arrays of the matrices A*P, A_diag*P_off and A_offd * P */
1233:   PetscCall(PetscMalloc1(am + 1, &api));
1234:   PetscCall(PetscMalloc1(am + 1, &adpoi));

1236:   adpoi[0]  = 0;
1237:   ptap->api = api;
1238:   api[0]    = 0;

1240:   /* create and initialize a linked list, will be used for both A_diag * P_loc_off and A_offd * P_oth */
1241:   PetscCall(PetscLLCondensedCreate(pN, pN, &lnk, &lnkbt));
1242:   MatPreallocateBegin(comm, am, pn, dnz, onz);

1244:   /* Symbolic calc of A_loc_diag * P_loc_diag */
1245:   PetscCall(MatGetOptionsPrefix(A, &prefix));
1246:   PetscCall(MatProductCreate(a->A, p->A, NULL, &adpd));
1247:   PetscCall(MatGetOptionsPrefix(A, &prefix));
1248:   PetscCall(MatSetOptionsPrefix(adpd, prefix));
1249:   PetscCall(MatAppendOptionsPrefix(adpd, "inner_diag_"));

1251:   PetscCall(MatProductSetType(adpd, MATPRODUCT_AB));
1252:   PetscCall(MatProductSetAlgorithm(adpd, "sorted"));
1253:   PetscCall(MatProductSetFill(adpd, fill));
1254:   PetscCall(MatProductSetFromOptions(adpd));

1256:   adpd->force_diagonals = C->force_diagonals;
1257:   PetscCall(MatProductSymbolic(adpd));

1259:   adpd_seq = (Mat_SeqAIJ *)adpd->data;
1260:   adpdi    = adpd_seq->i;
1261:   adpdj    = adpd_seq->j;
1262:   p_off    = (Mat_SeqAIJ *)p->B->data;
1263:   poff_i   = p_off->i;
1264:   poff_j   = p_off->j;

1266:   /* j_temp stores indices of a result row before they are added to the linked list */
1267:   PetscCall(PetscMalloc1(pN, &j_temp));

1269:   /* Symbolic calc of the A_diag * p_loc_off */
1270:   /* Initial FreeSpace size is fill*(nnz(A)+nnz(P)) */
1271:   PetscCall(PetscFreeSpaceGet(PetscRealIntMultTruncate(fill, PetscIntSumTruncate(adi[am], PetscIntSumTruncate(aoi[am], pi_loc[pm]))), &free_space_diag));
1272:   current_space = free_space_diag;

1274:   for (i = 0; i < am; i++) {
1275:     /* A_diag * P_loc_off */
1276:     nzi = adi[i + 1] - adi[i];
1277:     for (j = 0; j < nzi; j++) {
1278:       row  = *adj++;
1279:       pnz  = poff_i[row + 1] - poff_i[row];
1280:       Jptr = poff_j + poff_i[row];
1281:       for (i1 = 0; i1 < pnz; i1++) j_temp[i1] = p->garray[Jptr[i1]];
1282:       /* add non-zero cols of P into the sorted linked list lnk */
1283:       PetscCall(PetscLLCondensedAddSorted(pnz, j_temp, lnk, lnkbt));
1284:     }

1286:     adponz       = lnk[0];
1287:     adpoi[i + 1] = adpoi[i] + adponz;

1289:     /* if free space is not available, double the total space in the list */
1290:     if (current_space->local_remaining < adponz) {
1291:       PetscCall(PetscFreeSpaceGet(PetscIntSumTruncate(adponz, current_space->total_array_size), &current_space));
1292:       nspacedouble++;
1293:     }

1295:     /* Copy data into free space, then initialize lnk */
1296:     PetscCall(PetscLLCondensedClean(pN, adponz, current_space->array, lnk, lnkbt));

1298:     current_space->array += adponz;
1299:     current_space->local_used += adponz;
1300:     current_space->local_remaining -= adponz;
1301:   }

1303:   /* Symbolic calc of A_off * P_oth */
1304:   PetscCall(MatSetOptionsPrefix(a->B, prefix));
1305:   PetscCall(MatAppendOptionsPrefix(a->B, "inner_offdiag_"));
1306:   PetscCall(MatCreate(PETSC_COMM_SELF, &aopoth));
1307:   PetscCall(MatMatMultSymbolic_SeqAIJ_SeqAIJ(a->B, ptap->P_oth, fill, aopoth));
1308:   aopoth_seq = (Mat_SeqAIJ *)aopoth->data;
1309:   aopothi    = aopoth_seq->i;
1310:   aopothj    = aopoth_seq->j;

1312:   /* Allocate space for apj, adpj, aopj, ... */
1313:   /* destroy lists of free space and other temporary array(s) */

1315:   PetscCall(PetscMalloc1(aopothi[am] + adpoi[am] + adpdi[am], &ptap->apj));
1316:   PetscCall(PetscMalloc1(adpoi[am], &adpoj));

1318:   /* Copy from linked list to j-array */
1319:   PetscCall(PetscFreeSpaceContiguous(&free_space_diag, adpoj));
1320:   PetscCall(PetscLLDestroy(lnk, lnkbt));

1322:   adpoJ = adpoj;
1323:   adpdJ = adpdj;
1324:   aopJ  = aopothj;
1325:   apj   = ptap->apj;
1326:   apJ   = apj; /* still empty */

1328:   /* Merge j-arrays of A_off * P, A_diag * P_loc_off, and */
1329:   /* A_diag * P_loc_diag to get A*P */
1330:   for (i = 0; i < am; i++) {
1331:     aopnz  = aopothi[i + 1] - aopothi[i];
1332:     adponz = adpoi[i + 1] - adpoi[i];
1333:     adpdnz = adpdi[i + 1] - adpdi[i];

1335:     /* Correct indices from A_diag*P_diag */
1336:     for (i1 = 0; i1 < adpdnz; i1++) adpdJ[i1] += p_colstart;
1337:     /* Merge j-arrays of A_diag * P_loc_off and A_diag * P_loc_diag and A_off * P_oth */
1338:     Merge3SortedArrays(adponz, adpoJ, adpdnz, adpdJ, aopnz, aopJ, &apnz, apJ);
1339:     PetscCall(MatPreallocateSet(i + rstart, apnz, apJ, dnz, onz));

1341:     aopJ += aopnz;
1342:     adpoJ += adponz;
1343:     adpdJ += adpdnz;
1344:     apJ += apnz;
1345:     api[i + 1] = api[i] + apnz;
1346:   }

1348:   /* malloc apa to store dense row A[i,:]*P */
1349:   PetscCall(PetscCalloc1(pN, &ptap->apa));

1351:   /* create and assemble symbolic parallel matrix C */
1352:   PetscCall(MatSetSizes(C, am, pn, PETSC_DETERMINE, PETSC_DETERMINE));
1353:   PetscCall(MatSetBlockSizesFromMats(C, A, P));
1354:   PetscCall(MatGetType(A, &mtype));
1355:   PetscCall(MatSetType(C, mtype));
1356:   PetscCall(MatMPIAIJSetPreallocation(C, 0, dnz, 0, onz));
1357:   MatPreallocateEnd(dnz, onz);

1359:   PetscCall(MatSetValues_MPIAIJ_CopyFromCSRFormat_Symbolic(C, apj, api));
1360:   PetscCall(MatSetOption(C, MAT_NO_OFF_PROC_ENTRIES, PETSC_TRUE));
1361:   PetscCall(MatAssemblyBegin(C, MAT_FINAL_ASSEMBLY));
1362:   PetscCall(MatAssemblyEnd(C, MAT_FINAL_ASSEMBLY));
1363:   PetscCall(MatSetOption(C, MAT_NEW_NONZERO_LOCATION_ERR, PETSC_TRUE));

1365:   C->ops->matmultnumeric = MatMatMultNumeric_MPIAIJ_MPIAIJ_nonscalable;
1366:   C->ops->productnumeric = MatProductNumeric_AB;

1368:   /* attach the supporting struct to C for reuse */
1369:   C->product->data    = ptap;
1370:   C->product->destroy = MatProductCtxDestroy_MPIAIJ_MatMatMult;

1372:   /* set MatInfo */
1373:   afill = (PetscReal)api[am] / (adi[am] + aoi[am] + pi_loc[pm] + 1) + 1.e-5;
1374:   if (afill < 1.0) afill = 1.0;
1375:   C->info.mallocs           = nspacedouble;
1376:   C->info.fill_ratio_given  = fill;
1377:   C->info.fill_ratio_needed = afill;

1379:   if (PetscDefined(USE_INFO)) {
1380:     if (api[am]) {
1381:       PetscCall(PetscInfo(C, "Reallocs %" PetscInt_FMT "; Fill ratio: given %g needed %g.\n", nspacedouble, (double)fill, (double)afill));
1382:       PetscCall(PetscInfo(C, "Use MatMatMult(A,B,MatReuse,%g,&C) for best performance.;\n", (double)afill));
1383:     } else PetscCall(PetscInfo(C, "Empty matrix product\n"));
1384:   }

1386:   PetscCall(MatDestroy(&aopoth));
1387:   PetscCall(MatDestroy(&adpd));
1388:   PetscCall(PetscFree(j_temp));
1389:   PetscCall(PetscFree(adpoj));
1390:   PetscCall(PetscFree(adpoi));
1391:   PetscFunctionReturn(PETSC_SUCCESS);
1392: }

1394: /* This routine only works when scall=MAT_REUSE_MATRIX! */
1395: PetscErrorCode MatTransposeMatMultNumeric_MPIAIJ_MPIAIJ_matmatmult(Mat P, Mat A, Mat C)
1396: {
1397:   MatProductCtx_APMPI *ptap;
1398:   Mat                  Pt;

1400:   PetscFunctionBegin;
1401:   MatCheckProduct(C, 3);
1402:   ptap = (MatProductCtx_APMPI *)C->product->data;
1403:   PetscCheck(ptap, PetscObjectComm((PetscObject)C), PETSC_ERR_ARG_WRONGSTATE, "PtAP cannot be computed. Missing data");
1404:   PetscCheck(ptap->Pt, PetscObjectComm((PetscObject)C), PETSC_ERR_ARG_WRONGSTATE, "PtA cannot be reused. Do not call MatProductClear()");

1406:   Pt = ptap->Pt;
1407:   PetscCall(MatTransposeSetPrecursor(P, Pt));
1408:   PetscCall(MatTranspose(P, MAT_REUSE_MATRIX, &Pt));
1409:   PetscCall(MatMatMultNumeric_MPIAIJ_MPIAIJ(Pt, A, C));
1410:   PetscFunctionReturn(PETSC_SUCCESS);
1411: }

1413: /* This routine is modified from MatPtAPSymbolic_MPIAIJ_MPIAIJ() */
1414: PetscErrorCode MatTransposeMatMultSymbolic_MPIAIJ_MPIAIJ_nonscalable(Mat P, Mat A, PetscReal fill, Mat C)
1415: {
1416:   MatProductCtx_APMPI     *ptap;
1417:   Mat_MPIAIJ              *p = (Mat_MPIAIJ *)P->data;
1418:   MPI_Comm                 comm;
1419:   PetscMPIInt              size, rank;
1420:   PetscFreeSpaceList       free_space = NULL, current_space = NULL;
1421:   PetscInt                 pn = P->cmap->n, aN = A->cmap->N, an = A->cmap->n;
1422:   PetscInt                *lnk, i, k, rstart;
1423:   PetscBT                  lnkbt;
1424:   PetscMPIInt              tagi, tagj, *len_si, *len_s, *len_ri, nrecv, proc, nsend;
1425:   PETSC_UNUSED PetscMPIInt icompleted = 0;
1426:   PetscInt               **buf_rj, **buf_ri, **buf_ri_k, row, ncols, *cols;
1427:   PetscInt                 len, *dnz, *onz, *owners, nzi;
1428:   PetscInt                 nrows, *buf_s, *buf_si, *buf_si_i, **nextrow, **nextci;
1429:   MPI_Request             *swaits, *rwaits;
1430:   MPI_Status              *sstatus, rstatus;
1431:   PetscLayout              rowmap;
1432:   PetscInt                *owners_co, *coi, *coj; /* i and j array of (p->B)^T*A*P - used in the communication */
1433:   PetscMPIInt             *len_r, *id_r;          /* array of length of comm->size, store send/recv matrix values */
1434:   PetscInt                *Jptr, *prmap = p->garray, con, j, Crmax;
1435:   Mat_SeqAIJ              *a_loc, *c_loc, *c_oth;
1436:   PetscHMapI               ta;
1437:   MatType                  mtype;
1438:   const char              *prefix;

1440:   PetscFunctionBegin;
1441:   PetscCall(PetscObjectGetComm((PetscObject)A, &comm));
1442:   PetscCallMPI(MPI_Comm_size(comm, &size));
1443:   PetscCallMPI(MPI_Comm_rank(comm, &rank));

1445:   /* create symbolic parallel matrix C */
1446:   PetscCall(MatGetType(A, &mtype));
1447:   PetscCall(MatSetType(C, mtype));

1449:   C->ops->transposematmultnumeric = MatTransposeMatMultNumeric_MPIAIJ_MPIAIJ_nonscalable;

1451:   /* create struct MatProductCtx_APMPI and attached it to C later */
1452:   PetscCall(PetscNew(&ptap));

1454:   /* (0) compute Rd = Pd^T, Ro = Po^T  */
1455:   PetscCall(MatTranspose(p->A, MAT_INITIAL_MATRIX, &ptap->Rd));
1456:   PetscCall(MatTranspose(p->B, MAT_INITIAL_MATRIX, &ptap->Ro));

1458:   /* (1) compute symbolic A_loc */
1459:   PetscCall(MatMPIAIJGetLocalMat(A, MAT_INITIAL_MATRIX, &ptap->A_loc));

1461:   /* (2-1) compute symbolic C_oth = Ro*A_loc  */
1462:   PetscCall(MatGetOptionsPrefix(A, &prefix));
1463:   PetscCall(MatSetOptionsPrefix(ptap->Ro, prefix));
1464:   PetscCall(MatAppendOptionsPrefix(ptap->Ro, "inner_offdiag_"));
1465:   PetscCall(MatCreate(PETSC_COMM_SELF, &ptap->C_oth));
1466:   PetscCall(MatMatMultSymbolic_SeqAIJ_SeqAIJ(ptap->Ro, ptap->A_loc, fill, ptap->C_oth));

1468:   /* (3) send coj of C_oth to other processors  */
1469:   /* determine row ownership */
1470:   PetscCall(PetscLayoutCreate(comm, &rowmap));
1471:   rowmap->n  = pn;
1472:   rowmap->bs = 1;
1473:   PetscCall(PetscLayoutSetUp(rowmap));
1474:   owners = rowmap->range;

1476:   /* determine the number of messages to send, their lengths */
1477:   PetscCall(PetscMalloc4(size, &len_s, size, &len_si, size, &sstatus, size + 1, &owners_co));
1478:   PetscCall(PetscArrayzero(len_s, size));
1479:   PetscCall(PetscArrayzero(len_si, size));

1481:   c_oth = (Mat_SeqAIJ *)ptap->C_oth->data;
1482:   coi   = c_oth->i;
1483:   coj   = c_oth->j;
1484:   con   = ptap->C_oth->rmap->n;
1485:   proc  = 0;
1486:   for (i = 0; i < con; i++) {
1487:     while (prmap[i] >= owners[proc + 1]) proc++;
1488:     len_si[proc]++;                     /* num of rows in Co(=Pt*A) to be sent to [proc] */
1489:     len_s[proc] += coi[i + 1] - coi[i]; /* num of nonzeros in Co to be sent to [proc] */
1490:   }

1492:   len          = 0; /* max length of buf_si[], see (4) */
1493:   owners_co[0] = 0;
1494:   nsend        = 0;
1495:   for (proc = 0; proc < size; proc++) {
1496:     owners_co[proc + 1] = owners_co[proc] + len_si[proc];
1497:     if (len_s[proc]) {
1498:       nsend++;
1499:       len_si[proc] = 2 * (len_si[proc] + 1); /* length of buf_si to be sent to [proc] */
1500:       len += len_si[proc];
1501:     }
1502:   }

1504:   /* determine the number and length of messages to receive for coi and coj  */
1505:   PetscCall(PetscGatherNumberOfMessages(comm, NULL, len_s, &nrecv));
1506:   PetscCall(PetscGatherMessageLengths2(comm, nsend, nrecv, len_s, len_si, &id_r, &len_r, &len_ri));

1508:   /* post the Irecv and Isend of coj */
1509:   PetscCall(PetscCommGetNewTag(comm, &tagj));
1510:   PetscCall(PetscPostIrecvInt(comm, tagj, nrecv, id_r, len_r, &buf_rj, &rwaits));
1511:   PetscCall(PetscMalloc1(nsend, &swaits));
1512:   for (proc = 0, k = 0; proc < size; proc++) {
1513:     if (!len_s[proc]) continue;
1514:     i = owners_co[proc];
1515:     PetscCallMPI(MPIU_Isend(coj + coi[i], len_s[proc], MPIU_INT, proc, tagj, comm, swaits + k));
1516:     k++;
1517:   }

1519:   /* (2-2) compute symbolic C_loc = Rd*A_loc */
1520:   PetscCall(MatSetOptionsPrefix(ptap->Rd, prefix));
1521:   PetscCall(MatAppendOptionsPrefix(ptap->Rd, "inner_diag_"));
1522:   PetscCall(MatCreate(PETSC_COMM_SELF, &ptap->C_loc));
1523:   PetscCall(MatMatMultSymbolic_SeqAIJ_SeqAIJ(ptap->Rd, ptap->A_loc, fill, ptap->C_loc));
1524:   c_loc = (Mat_SeqAIJ *)ptap->C_loc->data;

1526:   /* receives coj are complete */
1527:   for (i = 0; i < nrecv; i++) PetscCallMPI(MPI_Waitany(nrecv, rwaits, &icompleted, &rstatus));
1528:   PetscCall(PetscFree(rwaits));
1529:   if (nsend) PetscCallMPI(MPI_Waitall(nsend, swaits, sstatus));

1531:   /* add received column indices into ta to update Crmax */
1532:   a_loc = (Mat_SeqAIJ *)ptap->A_loc->data;

1534:   /* create and initialize a linked list */
1535:   PetscCall(PetscHMapICreateWithSize(an, &ta)); /* for compute Crmax */
1536:   MatRowMergeMax_SeqAIJ(a_loc, ptap->A_loc->rmap->N, ta);

1538:   for (k = 0; k < nrecv; k++) { /* k-th received message */
1539:     Jptr = buf_rj[k];
1540:     for (j = 0; j < len_r[k]; j++) PetscCall(PetscHMapISet(ta, *(Jptr + j) + 1, 1));
1541:   }
1542:   PetscCall(PetscHMapIGetSize(ta, &Crmax));
1543:   PetscCall(PetscHMapIDestroy(&ta));

1545:   /* (4) send and recv coi */
1546:   PetscCall(PetscCommGetNewTag(comm, &tagi));
1547:   PetscCall(PetscPostIrecvInt(comm, tagi, nrecv, id_r, len_ri, &buf_ri, &rwaits));
1548:   PetscCall(PetscMalloc1(len, &buf_s));
1549:   buf_si = buf_s; /* points to the beginning of k-th msg to be sent */
1550:   for (proc = 0, k = 0; proc < size; proc++) {
1551:     if (!len_s[proc]) continue;
1552:     /* form outgoing message for i-structure:
1553:          buf_si[0]:                 nrows to be sent
1554:                [1:nrows]:           row index (global)
1555:                [nrows+1:2*nrows+1]: i-structure index
1556:     */
1557:     nrows       = len_si[proc] / 2 - 1; /* num of rows in Co to be sent to [proc] */
1558:     buf_si_i    = buf_si + nrows + 1;
1559:     buf_si[0]   = nrows;
1560:     buf_si_i[0] = 0;
1561:     nrows       = 0;
1562:     for (i = owners_co[proc]; i < owners_co[proc + 1]; i++) {
1563:       nzi                 = coi[i + 1] - coi[i];
1564:       buf_si_i[nrows + 1] = buf_si_i[nrows] + nzi;   /* i-structure */
1565:       buf_si[nrows + 1]   = prmap[i] - owners[proc]; /* local row index */
1566:       nrows++;
1567:     }
1568:     PetscCallMPI(MPIU_Isend(buf_si, len_si[proc], MPIU_INT, proc, tagi, comm, swaits + k));
1569:     k++;
1570:     buf_si += len_si[proc];
1571:   }
1572:   for (i = 0; i < nrecv; i++) PetscCallMPI(MPI_Waitany(nrecv, rwaits, &icompleted, &rstatus));
1573:   PetscCall(PetscFree(rwaits));
1574:   if (nsend) PetscCallMPI(MPI_Waitall(nsend, swaits, sstatus));

1576:   PetscCall(PetscFree4(len_s, len_si, sstatus, owners_co));
1577:   PetscCall(PetscFree(len_ri));
1578:   PetscCall(PetscFree(swaits));
1579:   PetscCall(PetscFree(buf_s));

1581:   /* (5) compute the local portion of C      */
1582:   /* set initial free space to be Crmax, sufficient for holding nonzeros in each row of C */
1583:   PetscCall(PetscFreeSpaceGet(Crmax, &free_space));
1584:   current_space = free_space;

1586:   PetscCall(PetscMalloc3(nrecv, &buf_ri_k, nrecv, &nextrow, nrecv, &nextci));
1587:   for (k = 0; k < nrecv; k++) {
1588:     buf_ri_k[k] = buf_ri[k]; /* beginning of k-th recved i-structure */
1589:     nrows       = *buf_ri_k[k];
1590:     nextrow[k]  = buf_ri_k[k] + 1;           /* next row number of k-th recved i-structure */
1591:     nextci[k]   = buf_ri_k[k] + (nrows + 1); /* points to the next i-structure of k-th recved i-structure  */
1592:   }

1594:   MatPreallocateBegin(comm, pn, an, dnz, onz);
1595:   PetscCall(PetscLLCondensedCreate(Crmax, aN, &lnk, &lnkbt));
1596:   for (i = 0; i < pn; i++) { /* for each local row of C */
1597:     /* add C_loc into C */
1598:     nzi  = c_loc->i[i + 1] - c_loc->i[i];
1599:     Jptr = c_loc->j + c_loc->i[i];
1600:     PetscCall(PetscLLCondensedAddSorted(nzi, Jptr, lnk, lnkbt));

1602:     /* add received col data into lnk */
1603:     for (k = 0; k < nrecv; k++) { /* k-th received message */
1604:       if (i == *nextrow[k]) {     /* i-th row */
1605:         nzi  = *(nextci[k] + 1) - *nextci[k];
1606:         Jptr = buf_rj[k] + *nextci[k];
1607:         PetscCall(PetscLLCondensedAddSorted(nzi, Jptr, lnk, lnkbt));
1608:         nextrow[k]++;
1609:         nextci[k]++;
1610:       }
1611:     }

1613:     /* add missing diagonal entry */
1614:     if (C->force_diagonals) {
1615:       k = i + owners[rank]; /* column index */
1616:       PetscCall(PetscLLCondensedAddSorted(1, &k, lnk, lnkbt));
1617:     }

1619:     nzi = lnk[0];

1621:     /* copy data into free space, then initialize lnk */
1622:     PetscCall(PetscLLCondensedClean(aN, nzi, current_space->array, lnk, lnkbt));
1623:     PetscCall(MatPreallocateSet(i + owners[rank], nzi, current_space->array, dnz, onz));
1624:   }
1625:   PetscCall(PetscFree3(buf_ri_k, nextrow, nextci));
1626:   PetscCall(PetscLLDestroy(lnk, lnkbt));
1627:   PetscCall(PetscFreeSpaceDestroy(free_space));

1629:   /* local sizes and preallocation */
1630:   PetscCall(MatSetSizes(C, pn, an, PETSC_DETERMINE, PETSC_DETERMINE));
1631:   PetscCall(PetscLayoutSetBlockSize(C->rmap, P->cmap->bs));
1632:   PetscCall(PetscLayoutSetBlockSize(C->cmap, A->cmap->bs));
1633:   PetscCall(MatMPIAIJSetPreallocation(C, 0, dnz, 0, onz));
1634:   MatPreallocateEnd(dnz, onz);

1636:   /* add C_loc and C_oth to C */
1637:   PetscCall(MatGetOwnershipRange(C, &rstart, NULL));
1638:   for (i = 0; i < pn; i++) {
1639:     ncols = c_loc->i[i + 1] - c_loc->i[i];
1640:     cols  = c_loc->j + c_loc->i[i];
1641:     row   = rstart + i;
1642:     PetscCall(MatSetValues(C, 1, (const PetscInt *)&row, ncols, (const PetscInt *)cols, NULL, INSERT_VALUES));

1644:     if (C->force_diagonals) PetscCall(MatSetValues(C, 1, (const PetscInt *)&row, 1, (const PetscInt *)&row, NULL, INSERT_VALUES));
1645:   }
1646:   for (i = 0; i < con; i++) {
1647:     ncols = c_oth->i[i + 1] - c_oth->i[i];
1648:     cols  = c_oth->j + c_oth->i[i];
1649:     row   = prmap[i];
1650:     PetscCall(MatSetValues(C, 1, (const PetscInt *)&row, ncols, (const PetscInt *)cols, NULL, INSERT_VALUES));
1651:   }
1652:   PetscCall(MatAssemblyBegin(C, MAT_FINAL_ASSEMBLY));
1653:   PetscCall(MatAssemblyEnd(C, MAT_FINAL_ASSEMBLY));
1654:   PetscCall(MatSetOption(C, MAT_NEW_NONZERO_LOCATION_ERR, PETSC_TRUE));

1656:   /* members in merge */
1657:   PetscCall(PetscFree(id_r));
1658:   PetscCall(PetscFree(len_r));
1659:   PetscCall(PetscFree(buf_ri[0]));
1660:   PetscCall(PetscFree(buf_ri));
1661:   PetscCall(PetscFree(buf_rj[0]));
1662:   PetscCall(PetscFree(buf_rj));
1663:   PetscCall(PetscLayoutDestroy(&rowmap));

1665:   /* attach the supporting struct to C for reuse */
1666:   C->product->data    = ptap;
1667:   C->product->destroy = MatProductCtxDestroy_MPIAIJ_PtAP;
1668:   PetscFunctionReturn(PETSC_SUCCESS);
1669: }

1671: PetscErrorCode MatTransposeMatMultNumeric_MPIAIJ_MPIAIJ_nonscalable(Mat P, Mat A, Mat C)
1672: {
1673:   Mat_MPIAIJ          *p = (Mat_MPIAIJ *)P->data;
1674:   Mat_SeqAIJ          *c_seq;
1675:   MatProductCtx_APMPI *ptap;
1676:   Mat                  A_loc, C_loc, C_oth;
1677:   PetscInt             i, rstart, rend, cm, ncols, row;
1678:   const PetscInt      *cols;
1679:   const PetscScalar   *vals;

1681:   PetscFunctionBegin;
1682:   MatCheckProduct(C, 3);
1683:   ptap = (MatProductCtx_APMPI *)C->product->data;
1684:   PetscCheck(ptap, PetscObjectComm((PetscObject)C), PETSC_ERR_ARG_WRONGSTATE, "PtAP cannot be computed. Missing data");
1685:   PetscCheck(ptap->A_loc, PetscObjectComm((PetscObject)C), PETSC_ERR_ARG_WRONGSTATE, "PtA cannot be reused. Do not call MatProductClear()");
1686:   PetscCall(MatZeroEntries(C));

1688:   /* These matrices are obtained in MatTransposeMatMultSymbolic() */
1689:   /* 1) get R = Pd^T, Ro = Po^T */
1690:   PetscCall(MatTransposeSetPrecursor(p->A, ptap->Rd));
1691:   PetscCall(MatTranspose(p->A, MAT_REUSE_MATRIX, &ptap->Rd));
1692:   PetscCall(MatTransposeSetPrecursor(p->B, ptap->Ro));
1693:   PetscCall(MatTranspose(p->B, MAT_REUSE_MATRIX, &ptap->Ro));

1695:   /* 2) compute numeric A_loc */
1696:   PetscCall(MatMPIAIJGetLocalMat(A, MAT_REUSE_MATRIX, &ptap->A_loc));

1698:   /* 3) C_loc = Rd*A_loc, C_oth = Ro*A_loc */
1699:   A_loc = ptap->A_loc;
1700:   PetscCall(ptap->C_loc->ops->matmultnumeric(ptap->Rd, A_loc, ptap->C_loc));
1701:   PetscCall(ptap->C_oth->ops->matmultnumeric(ptap->Ro, A_loc, ptap->C_oth));
1702:   C_loc = ptap->C_loc;
1703:   C_oth = ptap->C_oth;

1705:   /* add C_loc and C_oth to C */
1706:   PetscCall(MatGetOwnershipRange(C, &rstart, &rend));

1708:   /* C_loc -> C */
1709:   cm    = C_loc->rmap->N;
1710:   c_seq = (Mat_SeqAIJ *)C_loc->data;
1711:   cols  = c_seq->j;
1712:   vals  = c_seq->a;
1713:   for (i = 0; i < cm; i++) {
1714:     ncols = c_seq->i[i + 1] - c_seq->i[i];
1715:     row   = rstart + i;
1716:     PetscCall(MatSetValues(C, 1, &row, ncols, cols, vals, ADD_VALUES));
1717:     cols += ncols;
1718:     vals += ncols;
1719:   }

1721:   /* Co -> C, off-processor part */
1722:   cm    = C_oth->rmap->N;
1723:   c_seq = (Mat_SeqAIJ *)C_oth->data;
1724:   cols  = c_seq->j;
1725:   vals  = c_seq->a;
1726:   for (i = 0; i < cm; i++) {
1727:     ncols = c_seq->i[i + 1] - c_seq->i[i];
1728:     row   = p->garray[i];
1729:     PetscCall(MatSetValues(C, 1, &row, ncols, cols, vals, ADD_VALUES));
1730:     cols += ncols;
1731:     vals += ncols;
1732:   }
1733:   PetscCall(MatAssemblyBegin(C, MAT_FINAL_ASSEMBLY));
1734:   PetscCall(MatAssemblyEnd(C, MAT_FINAL_ASSEMBLY));
1735:   PetscCall(MatSetOption(C, MAT_NEW_NONZERO_LOCATION_ERR, PETSC_TRUE));
1736:   PetscFunctionReturn(PETSC_SUCCESS);
1737: }

1739: PetscErrorCode MatTransposeMatMultNumeric_MPIAIJ_MPIAIJ(Mat P, Mat A, Mat C)
1740: {
1741:   MatMergeSeqsToMPI   *merge;
1742:   Mat_MPIAIJ          *p  = (Mat_MPIAIJ *)P->data;
1743:   Mat_SeqAIJ          *pd = (Mat_SeqAIJ *)p->A->data, *po = (Mat_SeqAIJ *)p->B->data;
1744:   MatProductCtx_APMPI *ap;
1745:   PetscInt            *adj;
1746:   PetscInt             i, j, k, anz, pnz, row, *cj, nexta;
1747:   MatScalar           *ada, *ca, valtmp;
1748:   PetscInt             am = A->rmap->n, cm = C->rmap->n, pon = p->B->cmap->n;
1749:   MPI_Comm             comm;
1750:   PetscMPIInt          size, rank, taga, *len_s, proc;
1751:   PetscInt            *owners, nrows, **buf_ri_k, **nextrow, **nextci;
1752:   PetscInt           **buf_ri, **buf_rj;
1753:   PetscInt             cnz = 0, *bj_i, *bi, *bj, bnz, nextcj; /* bi,bj,ba: local array of C(mpi mat) */
1754:   MPI_Request         *s_waits, *r_waits;
1755:   MPI_Status          *status;
1756:   MatScalar          **abuf_r, *ba_i, *pA, *coa, *ba;
1757:   const PetscScalar   *dummy;
1758:   PetscInt            *ai, *aj, *coi, *coj, *poJ, *pdJ;
1759:   Mat                  A_loc;
1760:   Mat_SeqAIJ          *a_loc;

1762:   PetscFunctionBegin;
1763:   MatCheckProduct(C, 3);
1764:   ap = (MatProductCtx_APMPI *)C->product->data;
1765:   PetscCheck(ap, PetscObjectComm((PetscObject)C), PETSC_ERR_ARG_WRONGSTATE, "PtA cannot be computed. Missing data");
1766:   PetscCheck(ap->A_loc, PetscObjectComm((PetscObject)C), PETSC_ERR_ARG_WRONGSTATE, "PtA cannot be reused. Do not call MatProductClear()");
1767:   PetscCall(PetscObjectGetComm((PetscObject)C, &comm));
1768:   PetscCallMPI(MPI_Comm_size(comm, &size));
1769:   PetscCallMPI(MPI_Comm_rank(comm, &rank));

1771:   merge = ap->merge;

1773:   /* 2) compute numeric C_seq = P_loc^T*A_loc */
1774:   /* get data from symbolic products */
1775:   coi = merge->coi;
1776:   coj = merge->coj;
1777:   PetscCall(PetscCalloc1(coi[pon], &coa));
1778:   bi     = merge->bi;
1779:   bj     = merge->bj;
1780:   owners = merge->rowmap->range;
1781:   PetscCall(PetscCalloc1(bi[cm], &ba));

1783:   /* get A_loc by taking all local rows of A */
1784:   A_loc = ap->A_loc;
1785:   PetscCall(MatMPIAIJGetLocalMat(A, MAT_REUSE_MATRIX, &A_loc));
1786:   a_loc = (Mat_SeqAIJ *)A_loc->data;
1787:   ai    = a_loc->i;
1788:   aj    = a_loc->j;

1790:   /* trigger copy to CPU */
1791:   PetscCall(MatSeqAIJGetArrayRead(p->A, &dummy));
1792:   PetscCall(MatSeqAIJRestoreArrayRead(p->A, &dummy));
1793:   PetscCall(MatSeqAIJGetArrayRead(p->B, &dummy));
1794:   PetscCall(MatSeqAIJRestoreArrayRead(p->B, &dummy));
1795:   for (i = 0; i < am; i++) {
1796:     anz = ai[i + 1] - ai[i];
1797:     adj = aj + ai[i];
1798:     ada = a_loc->a + ai[i];

1800:     /* 2-b) Compute Cseq = P_loc[i,:]^T*A[i,:] using outer product */
1801:     /* put the value into Co=(p->B)^T*A (off-diagonal part, send to others) */
1802:     pnz = po->i[i + 1] - po->i[i];
1803:     poJ = po->j + po->i[i];
1804:     pA  = po->a + po->i[i];
1805:     for (j = 0; j < pnz; j++) {
1806:       row = poJ[j];
1807:       cj  = coj + coi[row];
1808:       ca  = coa + coi[row];
1809:       /* perform sparse axpy */
1810:       nexta  = 0;
1811:       valtmp = pA[j];
1812:       for (k = 0; nexta < anz; k++) {
1813:         if (cj[k] == adj[nexta]) {
1814:           ca[k] += valtmp * ada[nexta];
1815:           nexta++;
1816:         }
1817:       }
1818:       PetscCall(PetscLogFlops(2.0 * anz));
1819:     }

1821:     /* put the value into Cd (diagonal part) */
1822:     pnz = pd->i[i + 1] - pd->i[i];
1823:     pdJ = pd->j + pd->i[i];
1824:     pA  = pd->a + pd->i[i];
1825:     for (j = 0; j < pnz; j++) {
1826:       row = pdJ[j];
1827:       cj  = bj + bi[row];
1828:       ca  = ba + bi[row];
1829:       /* perform sparse axpy */
1830:       nexta  = 0;
1831:       valtmp = pA[j];
1832:       for (k = 0; nexta < anz; k++) {
1833:         if (cj[k] == adj[nexta]) {
1834:           ca[k] += valtmp * ada[nexta];
1835:           nexta++;
1836:         }
1837:       }
1838:       PetscCall(PetscLogFlops(2.0 * anz));
1839:     }
1840:   }

1842:   /* 3) send and recv matrix values coa */
1843:   buf_ri = merge->buf_ri;
1844:   buf_rj = merge->buf_rj;
1845:   len_s  = merge->len_s;
1846:   PetscCall(PetscCommGetNewTag(comm, &taga));
1847:   PetscCall(PetscPostIrecvScalar(comm, taga, merge->nrecv, merge->id_r, merge->len_r, &abuf_r, &r_waits));

1849:   PetscCall(PetscMalloc2(merge->nsend, &s_waits, size, &status));
1850:   for (proc = 0, k = 0; proc < size; proc++) {
1851:     if (!len_s[proc]) continue;
1852:     i = merge->owners_co[proc];
1853:     PetscCallMPI(MPIU_Isend(coa + coi[i], len_s[proc], MPIU_MATSCALAR, proc, taga, comm, s_waits + k));
1854:     k++;
1855:   }
1856:   if (merge->nrecv) PetscCallMPI(MPI_Waitall(merge->nrecv, r_waits, status));
1857:   if (merge->nsend) PetscCallMPI(MPI_Waitall(merge->nsend, s_waits, status));

1859:   PetscCall(PetscFree2(s_waits, status));
1860:   PetscCall(PetscFree(r_waits));
1861:   PetscCall(PetscFree(coa));

1863:   /* 4) insert local Cseq and received values into Cmpi */
1864:   PetscCall(PetscMalloc3(merge->nrecv, &buf_ri_k, merge->nrecv, &nextrow, merge->nrecv, &nextci));
1865:   for (k = 0; k < merge->nrecv; k++) {
1866:     buf_ri_k[k] = buf_ri[k]; /* beginning of k-th recved i-structure */
1867:     nrows       = *buf_ri_k[k];
1868:     nextrow[k]  = buf_ri_k[k] + 1;           /* next row number of k-th recved i-structure */
1869:     nextci[k]   = buf_ri_k[k] + (nrows + 1); /* points to the next i-structure of k-th recved i-structure  */
1870:   }

1872:   for (i = 0; i < cm; i++) {
1873:     row  = owners[rank] + i; /* global row index of C_seq */
1874:     bj_i = bj + bi[i];       /* col indices of the i-th row of C */
1875:     ba_i = ba + bi[i];
1876:     bnz  = bi[i + 1] - bi[i];
1877:     /* add received vals into ba */
1878:     for (k = 0; k < merge->nrecv; k++) { /* k-th received message */
1879:       /* i-th row */
1880:       if (i == *nextrow[k]) {
1881:         cnz    = *(nextci[k] + 1) - *nextci[k];
1882:         cj     = buf_rj[k] + *nextci[k];
1883:         ca     = abuf_r[k] + *nextci[k];
1884:         nextcj = 0;
1885:         for (j = 0; nextcj < cnz; j++) {
1886:           if (bj_i[j] == cj[nextcj]) { /* bcol == ccol */
1887:             ba_i[j] += ca[nextcj++];
1888:           }
1889:         }
1890:         nextrow[k]++;
1891:         nextci[k]++;
1892:         PetscCall(PetscLogFlops(2.0 * cnz));
1893:       }
1894:     }
1895:     PetscCall(MatSetValues(C, 1, &row, bnz, bj_i, ba_i, INSERT_VALUES));
1896:   }
1897:   PetscCall(MatAssemblyBegin(C, MAT_FINAL_ASSEMBLY));
1898:   PetscCall(MatAssemblyEnd(C, MAT_FINAL_ASSEMBLY));

1900:   PetscCall(PetscFree(ba));
1901:   PetscCall(PetscFree(abuf_r[0]));
1902:   PetscCall(PetscFree(abuf_r));
1903:   PetscCall(PetscFree3(buf_ri_k, nextrow, nextci));
1904:   PetscFunctionReturn(PETSC_SUCCESS);
1905: }

1907: PetscErrorCode MatTransposeMatMultSymbolic_MPIAIJ_MPIAIJ(Mat P, Mat A, PetscReal fill, Mat C)
1908: {
1909:   Mat                  A_loc;
1910:   MatProductCtx_APMPI *ap;
1911:   PetscFreeSpaceList   free_space = NULL, current_space = NULL;
1912:   Mat_MPIAIJ          *p = (Mat_MPIAIJ *)P->data, *a = (Mat_MPIAIJ *)A->data;
1913:   PetscInt            *pdti, *pdtj, *poti, *potj, *ptJ;
1914:   PetscInt             nnz;
1915:   PetscInt            *lnk, *owners_co, *coi, *coj, i, k, pnz, row;
1916:   PetscInt             am = A->rmap->n, pn = P->cmap->n;
1917:   MPI_Comm             comm;
1918:   PetscMPIInt          size, rank, tagi, tagj, *len_si, *len_s, *len_ri, proc;
1919:   PetscInt           **buf_rj, **buf_ri, **buf_ri_k;
1920:   PetscInt             len, *dnz, *onz, *owners;
1921:   PetscInt             nzi, *bi, *bj;
1922:   PetscInt             nrows, *buf_s, *buf_si, *buf_si_i, **nextrow, **nextci;
1923:   MPI_Request         *swaits, *rwaits;
1924:   MPI_Status          *sstatus, rstatus;
1925:   MatMergeSeqsToMPI   *merge;
1926:   PetscInt            *ai, *aj, *Jptr, anz, *prmap = p->garray, pon, nspacedouble = 0, j;
1927:   PetscReal            afill  = 1.0, afill_tmp;
1928:   PetscInt             rstart = P->cmap->rstart, rmax, Armax;
1929:   Mat_SeqAIJ          *a_loc;
1930:   PetscHMapI           ta;
1931:   MatType              mtype;

1933:   PetscFunctionBegin;
1934:   PetscCall(PetscObjectGetComm((PetscObject)A, &comm));
1935:   /* check if matrix local sizes are compatible */
1936:   PetscCheck(A->rmap->rstart == P->rmap->rstart && A->rmap->rend == P->rmap->rend, comm, PETSC_ERR_ARG_SIZ, "Matrix local dimensions are incompatible, A (%" PetscInt_FMT ", %" PetscInt_FMT ") != P (%" PetscInt_FMT ",%" PetscInt_FMT ")", A->rmap->rstart,
1937:              A->rmap->rend, P->rmap->rstart, P->rmap->rend);

1939:   PetscCallMPI(MPI_Comm_size(comm, &size));
1940:   PetscCallMPI(MPI_Comm_rank(comm, &rank));

1942:   /* create struct MatProductCtx_APMPI and attached it to C later */
1943:   PetscCall(PetscNew(&ap));

1945:   /* get A_loc by taking all local rows of A */
1946:   PetscCall(MatMPIAIJGetLocalMat_Private(A, MAT_INITIAL_MATRIX, C->structure_only, &A_loc));

1948:   ap->A_loc = A_loc;
1949:   a_loc     = (Mat_SeqAIJ *)A_loc->data;
1950:   ai        = a_loc->i;
1951:   aj        = a_loc->j;

1953:   /* determine symbolic Co=(p->B)^T*A - send to others */
1954:   PetscCall(MatGetSymbolicTranspose_SeqAIJ(p->A, &pdti, &pdtj));
1955:   PetscCall(MatGetSymbolicTranspose_SeqAIJ(p->B, &poti, &potj));
1956:   pon = p->B->cmap->n; /* total num of rows to be sent to other processors
1957:                          >= (num of nonzero rows of C_seq) - pn */
1958:   PetscCall(PetscMalloc1(pon + 1, &coi));
1959:   coi[0] = 0;

1961:   /* set initial free space to be fill*(nnz(p->B) + nnz(A)) */
1962:   nnz = PetscRealIntMultTruncate(fill, PetscIntSumTruncate(poti[pon], ai[am]));
1963:   PetscCall(PetscFreeSpaceGet(nnz, &free_space));
1964:   current_space = free_space;

1966:   /* create and initialize a linked list */
1967:   PetscCall(PetscHMapICreateWithSize(A->cmap->n + a->B->cmap->N, &ta));
1968:   MatRowMergeMax_SeqAIJ(a_loc, am, ta);
1969:   PetscCall(PetscHMapIGetSize(ta, &Armax));

1971:   PetscCall(PetscLLCondensedCreate_Scalable(Armax, &lnk));

1973:   for (i = 0; i < pon; i++) {
1974:     pnz = poti[i + 1] - poti[i];
1975:     ptJ = potj + poti[i];
1976:     for (j = 0; j < pnz; j++) {
1977:       row  = ptJ[j]; /* row of A_loc == col of Pot */
1978:       anz  = ai[row + 1] - ai[row];
1979:       Jptr = aj + ai[row];
1980:       /* add non-zero cols of AP into the sorted linked list lnk */
1981:       PetscCall(PetscLLCondensedAddSorted_Scalable(anz, Jptr, lnk));
1982:     }
1983:     nnz = lnk[0];

1985:     /* If free space is not available, double the total space in the list */
1986:     if (current_space->local_remaining < nnz) {
1987:       PetscCall(PetscFreeSpaceGet(PetscIntSumTruncate(nnz, current_space->total_array_size), &current_space));
1988:       nspacedouble++;
1989:     }

1991:     /* Copy data into free space, and zero out denserows */
1992:     PetscCall(PetscLLCondensedClean_Scalable(nnz, current_space->array, lnk));

1994:     current_space->array += nnz;
1995:     current_space->local_used += nnz;
1996:     current_space->local_remaining -= nnz;

1998:     coi[i + 1] = coi[i] + nnz;
1999:   }

2001:   PetscCall(PetscMalloc1(coi[pon], &coj));
2002:   PetscCall(PetscFreeSpaceContiguous(&free_space, coj));
2003:   PetscCall(PetscLLCondensedDestroy_Scalable(lnk)); /* must destroy to get a new one for C */

2005:   afill_tmp = (PetscReal)coi[pon] / (poti[pon] + ai[am] + 1);
2006:   if (afill_tmp > afill) afill = afill_tmp;

2008:   /* send j-array (coj) of Co to other processors */
2009:   /* determine row ownership */
2010:   PetscCall(PetscNew(&merge));
2011:   PetscCall(PetscLayoutCreate(comm, &merge->rowmap));

2013:   merge->rowmap->n  = pn;
2014:   merge->rowmap->bs = 1;

2016:   PetscCall(PetscLayoutSetUp(merge->rowmap));
2017:   owners = merge->rowmap->range;

2019:   /* determine the number of messages to send, their lengths */
2020:   PetscCall(PetscCalloc1(size, &len_si));
2021:   PetscCall(PetscCalloc1(size, &merge->len_s));

2023:   len_s        = merge->len_s;
2024:   merge->nsend = 0;

2026:   PetscCall(PetscMalloc1(size + 1, &owners_co));

2028:   proc = 0;
2029:   for (i = 0; i < pon; i++) {
2030:     while (prmap[i] >= owners[proc + 1]) proc++;
2031:     len_si[proc]++; /* num of rows in Co to be sent to [proc] */
2032:     len_s[proc] += coi[i + 1] - coi[i];
2033:   }

2035:   len          = 0; /* max length of buf_si[] */
2036:   owners_co[0] = 0;
2037:   for (proc = 0; proc < size; proc++) {
2038:     owners_co[proc + 1] = owners_co[proc] + len_si[proc];
2039:     if (len_s[proc]) {
2040:       merge->nsend++;
2041:       len_si[proc] = 2 * (len_si[proc] + 1);
2042:       len += len_si[proc];
2043:     }
2044:   }

2046:   /* determine the number and length of messages to receive for coi and coj  */
2047:   PetscCall(PetscGatherNumberOfMessages(comm, NULL, len_s, &merge->nrecv));
2048:   PetscCall(PetscGatherMessageLengths2(comm, merge->nsend, merge->nrecv, len_s, len_si, &merge->id_r, &merge->len_r, &len_ri));

2050:   /* post the Irecv and Isend of coj */
2051:   PetscCall(PetscCommGetNewTag(comm, &tagj));
2052:   PetscCall(PetscPostIrecvInt(comm, tagj, merge->nrecv, merge->id_r, merge->len_r, &buf_rj, &rwaits));
2053:   PetscCall(PetscMalloc1(merge->nsend, &swaits));
2054:   for (proc = 0, k = 0; proc < size; proc++) {
2055:     if (!len_s[proc]) continue;
2056:     i = owners_co[proc];
2057:     PetscCallMPI(MPIU_Isend(coj + coi[i], len_s[proc], MPIU_INT, proc, tagj, comm, swaits + k));
2058:     k++;
2059:   }

2061:   /* receives and sends of coj are complete */
2062:   PetscCall(PetscMalloc1(size, &sstatus));
2063:   for (i = 0; i < merge->nrecv; i++) {
2064:     PETSC_UNUSED PetscMPIInt icompleted;
2065:     PetscCallMPI(MPI_Waitany(merge->nrecv, rwaits, &icompleted, &rstatus));
2066:   }
2067:   PetscCall(PetscFree(rwaits));
2068:   if (merge->nsend) PetscCallMPI(MPI_Waitall(merge->nsend, swaits, sstatus));

2070:   /* add received column indices into table to update Armax */
2071:   /* Armax can be as large as aN if a P[row,:] is dense, see src/ksp/ksp/tutorials/ex56.c! */
2072:   for (k = 0; k < merge->nrecv; k++) { /* k-th received message */
2073:     Jptr = buf_rj[k];
2074:     for (j = 0; j < merge->len_r[k]; j++) PetscCall(PetscHMapISet(ta, *(Jptr + j) + 1, 1));
2075:   }
2076:   PetscCall(PetscHMapIGetSize(ta, &Armax));

2078:   /* send and recv coi */
2079:   PetscCall(PetscCommGetNewTag(comm, &tagi));
2080:   PetscCall(PetscPostIrecvInt(comm, tagi, merge->nrecv, merge->id_r, len_ri, &buf_ri, &rwaits));
2081:   PetscCall(PetscMalloc1(len, &buf_s));
2082:   buf_si = buf_s; /* points to the beginning of k-th msg to be sent */
2083:   for (proc = 0, k = 0; proc < size; proc++) {
2084:     if (!len_s[proc]) continue;
2085:     /* form outgoing message for i-structure:
2086:          buf_si[0]:                 nrows to be sent
2087:                [1:nrows]:           row index (global)
2088:                [nrows+1:2*nrows+1]: i-structure index
2089:     */
2090:     nrows       = len_si[proc] / 2 - 1;
2091:     buf_si_i    = buf_si + nrows + 1;
2092:     buf_si[0]   = nrows;
2093:     buf_si_i[0] = 0;
2094:     nrows       = 0;
2095:     for (i = owners_co[proc]; i < owners_co[proc + 1]; i++) {
2096:       nzi                 = coi[i + 1] - coi[i];
2097:       buf_si_i[nrows + 1] = buf_si_i[nrows] + nzi;   /* i-structure */
2098:       buf_si[nrows + 1]   = prmap[i] - owners[proc]; /* local row index */
2099:       nrows++;
2100:     }
2101:     PetscCallMPI(MPIU_Isend(buf_si, len_si[proc], MPIU_INT, proc, tagi, comm, swaits + k));
2102:     k++;
2103:     buf_si += len_si[proc];
2104:   }
2105:   i = merge->nrecv;
2106:   while (i--) {
2107:     PETSC_UNUSED PetscMPIInt icompleted;
2108:     PetscCallMPI(MPI_Waitany(merge->nrecv, rwaits, &icompleted, &rstatus));
2109:   }
2110:   PetscCall(PetscFree(rwaits));
2111:   if (merge->nsend) PetscCallMPI(MPI_Waitall(merge->nsend, swaits, sstatus));
2112:   PetscCall(PetscFree(len_si));
2113:   PetscCall(PetscFree(len_ri));
2114:   PetscCall(PetscFree(swaits));
2115:   PetscCall(PetscFree(sstatus));
2116:   PetscCall(PetscFree(buf_s));

2118:   /* compute the local portion of C (mpi mat) */
2119:   /* allocate bi array and free space for accumulating nonzero column info */
2120:   PetscCall(PetscMalloc1(pn + 1, &bi));
2121:   bi[0] = 0;

2123:   /* set initial free space to be fill*(nnz(P) + nnz(AP)) */
2124:   nnz = PetscRealIntMultTruncate(fill, PetscIntSumTruncate(pdti[pn], PetscIntSumTruncate(poti[pon], ai[am])));
2125:   PetscCall(PetscFreeSpaceGet(nnz, &free_space));
2126:   current_space = free_space;

2128:   PetscCall(PetscMalloc3(merge->nrecv, &buf_ri_k, merge->nrecv, &nextrow, merge->nrecv, &nextci));
2129:   for (k = 0; k < merge->nrecv; k++) {
2130:     buf_ri_k[k] = buf_ri[k]; /* beginning of k-th recved i-structure */
2131:     nrows       = *buf_ri_k[k];
2132:     nextrow[k]  = buf_ri_k[k] + 1;           /* next row number of k-th recved i-structure */
2133:     nextci[k]   = buf_ri_k[k] + (nrows + 1); /* points to the next i-structure of k-th received i-structure  */
2134:   }

2136:   PetscCall(PetscLLCondensedCreate_Scalable(Armax, &lnk));
2137:   MatPreallocateBegin(comm, pn, A->cmap->n, dnz, onz);
2138:   rmax = 0;
2139:   for (i = 0; i < pn; i++) {
2140:     /* add pdt[i,:]*AP into lnk */
2141:     pnz = pdti[i + 1] - pdti[i];
2142:     ptJ = pdtj + pdti[i];
2143:     for (j = 0; j < pnz; j++) {
2144:       row  = ptJ[j]; /* row of AP == col of Pt */
2145:       anz  = ai[row + 1] - ai[row];
2146:       Jptr = aj + ai[row];
2147:       /* add non-zero cols of AP into the sorted linked list lnk */
2148:       PetscCall(PetscLLCondensedAddSorted_Scalable(anz, Jptr, lnk));
2149:     }

2151:     /* add received col data into lnk */
2152:     for (k = 0; k < merge->nrecv; k++) { /* k-th received message */
2153:       if (i == *nextrow[k]) {            /* i-th row */
2154:         nzi  = *(nextci[k] + 1) - *nextci[k];
2155:         Jptr = buf_rj[k] + *nextci[k];
2156:         PetscCall(PetscLLCondensedAddSorted_Scalable(nzi, Jptr, lnk));
2157:         nextrow[k]++;
2158:         nextci[k]++;
2159:       }
2160:     }

2162:     /* add missing diagonal entry */
2163:     if (C->force_diagonals) {
2164:       k = i + owners[rank]; /* column index */
2165:       PetscCall(PetscLLCondensedAddSorted_Scalable(1, &k, lnk));
2166:     }

2168:     nnz = lnk[0];

2170:     /* if free space is not available, make more free space */
2171:     if (current_space->local_remaining < nnz) {
2172:       PetscCall(PetscFreeSpaceGet(PetscIntSumTruncate(nnz, current_space->total_array_size), &current_space));
2173:       nspacedouble++;
2174:     }
2175:     /* copy data into free space, then initialize lnk */
2176:     PetscCall(PetscLLCondensedClean_Scalable(nnz, current_space->array, lnk));
2177:     PetscCall(MatPreallocateSet(i + owners[rank], nnz, current_space->array, dnz, onz));

2179:     current_space->array += nnz;
2180:     current_space->local_used += nnz;
2181:     current_space->local_remaining -= nnz;

2183:     bi[i + 1] = bi[i] + nnz;
2184:     if (nnz > rmax) rmax = nnz;
2185:   }
2186:   PetscCall(PetscFree3(buf_ri_k, nextrow, nextci));

2188:   PetscCall(PetscMalloc1(bi[pn], &bj));
2189:   PetscCall(PetscFreeSpaceContiguous(&free_space, bj));
2190:   afill_tmp = (PetscReal)bi[pn] / (pdti[pn] + poti[pon] + ai[am] + 1);
2191:   if (afill_tmp > afill) afill = afill_tmp;
2192:   PetscCall(PetscLLCondensedDestroy_Scalable(lnk));
2193:   PetscCall(PetscHMapIDestroy(&ta));
2194:   PetscCall(MatRestoreSymbolicTranspose_SeqAIJ(p->A, &pdti, &pdtj));
2195:   PetscCall(MatRestoreSymbolicTranspose_SeqAIJ(p->B, &poti, &potj));

2197:   /* create symbolic parallel matrix C - why cannot be assembled in Numeric part   */
2198:   PetscCall(MatSetSizes(C, pn, A->cmap->n, PETSC_DETERMINE, PETSC_DETERMINE));
2199:   PetscCall(MatSetBlockSizes(C, P->cmap->bs, A->cmap->bs));
2200:   PetscCall(MatGetType(A, &mtype));
2201:   PetscCall(MatSetType(C, C->structure_only ? MATMPIAIJ : mtype));
2202:   PetscCall(MatMPIAIJSetPreallocation(C, 0, dnz, 0, onz));
2203:   MatPreallocateEnd(dnz, onz);
2204:   PetscCall(MatSetBlockSize(C, 1));
2205:   PetscCall(MatSetOption(C, MAT_NO_OFF_PROC_ENTRIES, PETSC_TRUE));
2206:   for (i = 0; i < pn; i++) {
2207:     row  = i + rstart;
2208:     nnz  = bi[i + 1] - bi[i];
2209:     Jptr = bj + bi[i];
2210:     PetscCall(MatSetValues(C, 1, &row, nnz, Jptr, NULL, INSERT_VALUES));
2211:   }
2212:   PetscCall(MatAssemblyBegin(C, MAT_FINAL_ASSEMBLY));
2213:   PetscCall(MatAssemblyEnd(C, MAT_FINAL_ASSEMBLY));
2214:   PetscCall(MatSetOption(C, MAT_NEW_NONZERO_LOCATION_ERR, PETSC_TRUE));
2215:   merge->bi        = bi;
2216:   merge->bj        = bj;
2217:   merge->coi       = coi;
2218:   merge->coj       = coj;
2219:   merge->buf_ri    = buf_ri;
2220:   merge->buf_rj    = buf_rj;
2221:   merge->owners_co = owners_co;

2223:   /* attach the supporting struct to C for reuse */
2224:   C->product->data    = ap;
2225:   C->product->destroy = MatProductCtxDestroy_MPIAIJ_PtAP;
2226:   ap->merge           = merge;

2228:   C->ops->mattransposemultnumeric = MatTransposeMatMultNumeric_MPIAIJ_MPIAIJ;

2230:   if (PetscDefined(USE_INFO)) {
2231:     if (bi[pn] != 0) {
2232:       PetscCall(PetscInfo(C, "Reallocs %" PetscInt_FMT "; Fill ratio: given %g needed %g.\n", nspacedouble, (double)fill, (double)afill));
2233:       PetscCall(PetscInfo(C, "Use MatTransposeMatMult(A,B,MatReuse,%g,&C) for best performance.\n", (double)afill));
2234:     } else PetscCall(PetscInfo(C, "Empty matrix product\n"));
2235:   }
2236:   PetscFunctionReturn(PETSC_SUCCESS);
2237: }

2239: static PetscErrorCode MatProductSymbolic_AtB_MPIAIJ_MPIAIJ(Mat C)
2240: {
2241:   Mat_Product *product = C->product;
2242:   Mat          A = product->A, B = product->B;
2243:   PetscReal    fill = product->fill;
2244:   PetscBool    flg;

2246:   PetscFunctionBegin;
2247:   /* scalable */
2248:   PetscCall(PetscStrcmp(product->alg, "scalable", &flg));
2249:   if (flg || C->structure_only) {
2250:     if (C->structure_only) PetscCall(MatProductSetAlgorithm(C, "scalable"));
2251:     PetscCall(MatTransposeMatMultSymbolic_MPIAIJ_MPIAIJ(A, B, fill, C));
2252:     goto next;
2253:   }

2255:   /* nonscalable */
2256:   PetscCall(PetscStrcmp(product->alg, "nonscalable", &flg));
2257:   if (flg) {
2258:     PetscCall(MatTransposeMatMultSymbolic_MPIAIJ_MPIAIJ_nonscalable(A, B, fill, C));
2259:     goto next;
2260:   }

2262:   /* matmatmult */
2263:   PetscCall(PetscStrcmp(product->alg, "at*b", &flg));
2264:   if (flg) {
2265:     Mat                  At;
2266:     MatProductCtx_APMPI *ptap;

2268:     PetscCall(MatTranspose(A, MAT_INITIAL_MATRIX, &At));
2269:     PetscCall(MatMatMultSymbolic_MPIAIJ_MPIAIJ(At, B, fill, C));
2270:     ptap = (MatProductCtx_APMPI *)C->product->data;
2271:     if (ptap) {
2272:       ptap->Pt            = At;
2273:       C->product->destroy = MatProductCtxDestroy_MPIAIJ_PtAP;
2274:     }
2275:     C->ops->transposematmultnumeric = MatTransposeMatMultNumeric_MPIAIJ_MPIAIJ_matmatmult;
2276:     goto next;
2277:   }

2279:   /* backend general code */
2280:   PetscCall(PetscStrcmp(product->alg, "backend", &flg));
2281:   if (flg) {
2282:     PetscCall(MatProductSymbolic_MPIAIJBACKEND(C));
2283:     PetscFunctionReturn(PETSC_SUCCESS);
2284:   }

2286:   SETERRQ(PETSC_COMM_SELF, PETSC_ERR_SUP, "MatProduct type is not supported");

2288: next:
2289:   C->ops->productnumeric = MatProductNumeric_AtB;
2290:   PetscFunctionReturn(PETSC_SUCCESS);
2291: }

2293: /* Set options for MatMatMultxxx_MPIAIJ_MPIAIJ */
2294: static PetscErrorCode MatProductSetFromOptions_MPIAIJ_AB(Mat C)
2295: {
2296:   Mat_Product *product = C->product;
2297:   Mat          A = product->A, B = product->B;
2298: #if PetscDefined(HAVE_HYPRE)
2299:   const char *algTypes[5] = {"scalable", "nonscalable", "seqmpi", "backend", "hypre"};
2300:   PetscInt    nalg        = 5;
2301: #else
2302:   const char *algTypes[4] = {
2303:     "scalable",
2304:     "nonscalable",
2305:     "seqmpi",
2306:     "backend",
2307:   };
2308:   PetscInt nalg = 4;
2309: #endif
2310:   PetscInt  alg = 1; /* set nonscalable algorithm as default */
2311:   PetscBool flg;
2312:   MPI_Comm  comm;

2314:   PetscFunctionBegin;
2315:   PetscCall(PetscObjectGetComm((PetscObject)C, &comm));

2317:   /* Set "nonscalable" as default algorithm */
2318:   PetscCall(PetscStrcmp(C->product->alg, "default", &flg));
2319:   if (flg) {
2320:     PetscCall(MatProductSetAlgorithm(C, algTypes[alg]));

2322:     /* Set "scalable" as default if BN and local nonzeros of A and B are large */
2323:     if (B->cmap->N > 100000) { /* may switch to scalable algorithm as default */
2324:       MatInfo   Ainfo, Binfo;
2325:       PetscInt  nz_local;
2326:       PetscBool alg_scalable = PETSC_FALSE;

2328:       PetscCall(MatGetInfo(A, MAT_LOCAL, &Ainfo));
2329:       PetscCall(MatGetInfo(B, MAT_LOCAL, &Binfo));
2330:       nz_local = (PetscInt)(Ainfo.nz_allocated + Binfo.nz_allocated);

2332:       if (B->cmap->N > product->fill * nz_local) alg_scalable = PETSC_TRUE;
2333:       PetscCallMPI(MPIU_Allreduce(MPI_IN_PLACE, &alg_scalable, 1, MPI_C_BOOL, MPI_LOR, comm));

2335:       if (alg_scalable) {
2336:         alg = 0; /* scalable algorithm would 50% slower than nonscalable algorithm */
2337:         PetscCall(MatProductSetAlgorithm(C, algTypes[alg]));
2338:         PetscCall(PetscInfo(B, "Use scalable algorithm, BN %" PetscInt_FMT ", fill*nz_allocated %g\n", B->cmap->N, (double)(product->fill * nz_local)));
2339:       }
2340:     }
2341:   }

2343:   /* Get runtime option */
2344:   if (product->api_user) {
2345:     PetscOptionsBegin(PetscObjectComm((PetscObject)C), ((PetscObject)C)->prefix, "MatMatMult", "Mat");
2346:     PetscCall(PetscOptionsEList("-matmatmult_via", "Algorithmic approach", "MatMatMult", algTypes, nalg, algTypes[alg], &alg, &flg));
2347:     PetscOptionsEnd();
2348:   } else {
2349:     PetscOptionsBegin(PetscObjectComm((PetscObject)C), ((PetscObject)C)->prefix, "MatProduct_AB", "Mat");
2350:     PetscCall(PetscOptionsEList("-mat_product_algorithm", "Algorithmic approach", "MatMatMult", algTypes, nalg, algTypes[alg], &alg, &flg));
2351:     PetscOptionsEnd();
2352:   }
2353:   if (flg) PetscCall(MatProductSetAlgorithm(C, algTypes[alg]));

2355:   C->ops->productsymbolic = MatProductSymbolic_AB_MPIAIJ_MPIAIJ;
2356:   PetscFunctionReturn(PETSC_SUCCESS);
2357: }

2359: static PetscErrorCode MatProductSetFromOptions_MPIAIJ_ABt(Mat C)
2360: {
2361:   PetscFunctionBegin;
2362:   PetscCall(MatProductSetFromOptions_MPIAIJ_AB(C));
2363:   C->ops->productsymbolic = MatProductSymbolic_ABt_MPIAIJ_MPIAIJ;
2364:   PetscFunctionReturn(PETSC_SUCCESS);
2365: }

2367: /* Set options for MatTransposeMatMultXXX_MPIAIJ_MPIAIJ */
2368: static PetscErrorCode MatProductSetFromOptions_MPIAIJ_AtB(Mat C)
2369: {
2370:   Mat_Product *product = C->product;
2371:   Mat          A = product->A, B = product->B;
2372:   const char  *algTypes[4] = {"scalable", "nonscalable", "at*b", "backend"};
2373:   PetscInt     nalg        = 4;
2374:   PetscInt     alg         = 1; /* set default algorithm  */
2375:   PetscBool    flg;
2376:   MPI_Comm     comm;

2378:   PetscFunctionBegin;
2379:   /* Check matrix local sizes */
2380:   PetscCall(PetscObjectGetComm((PetscObject)C, &comm));
2381:   PetscCheck(A->rmap->rstart == B->rmap->rstart && A->rmap->rend == B->rmap->rend, PETSC_COMM_SELF, PETSC_ERR_ARG_SIZ, "Matrix local dimensions are incompatible, A (%" PetscInt_FMT ", %" PetscInt_FMT ") != B (%" PetscInt_FMT ",%" PetscInt_FMT ")",
2382:              A->rmap->rstart, A->rmap->rend, B->rmap->rstart, B->rmap->rend);

2384:   /* Set default algorithm */
2385:   PetscCall(PetscStrcmp(C->product->alg, "default", &flg));
2386:   if (flg) PetscCall(MatProductSetAlgorithm(C, algTypes[alg]));

2388:   /* Set "scalable" as default if BN and local nonzeros of A and B are large */
2389:   if (alg && B->cmap->N > 100000) { /* may switch to scalable algorithm as default */
2390:     MatInfo   Ainfo, Binfo;
2391:     PetscInt  nz_local;
2392:     PetscBool alg_scalable = PETSC_FALSE;

2394:     PetscCall(MatGetInfo(A, MAT_LOCAL, &Ainfo));
2395:     PetscCall(MatGetInfo(B, MAT_LOCAL, &Binfo));
2396:     nz_local = (PetscInt)(Ainfo.nz_allocated + Binfo.nz_allocated);

2398:     if (B->cmap->N > product->fill * nz_local) alg_scalable = PETSC_TRUE;
2399:     PetscCallMPI(MPIU_Allreduce(MPI_IN_PLACE, &alg_scalable, 1, MPI_C_BOOL, MPI_LOR, comm));

2401:     if (alg_scalable) {
2402:       alg = 0; /* scalable algorithm would 50% slower than nonscalable algorithm */
2403:       PetscCall(MatProductSetAlgorithm(C, algTypes[alg]));
2404:       PetscCall(PetscInfo(B, "Use scalable algorithm, BN %" PetscInt_FMT ", fill*nz_allocated %g\n", B->cmap->N, (double)(product->fill * nz_local)));
2405:     }
2406:   }

2408:   /* Get runtime option */
2409:   if (product->api_user) {
2410:     PetscOptionsBegin(PetscObjectComm((PetscObject)C), ((PetscObject)C)->prefix, "MatTransposeMatMult", "Mat");
2411:     PetscCall(PetscOptionsEList("-mattransposematmult_via", "Algorithmic approach", "MatTransposeMatMult", algTypes, nalg, algTypes[alg], &alg, &flg));
2412:     PetscOptionsEnd();
2413:   } else {
2414:     PetscOptionsBegin(PetscObjectComm((PetscObject)C), ((PetscObject)C)->prefix, "MatProduct_AtB", "Mat");
2415:     PetscCall(PetscOptionsEList("-mat_product_algorithm", "Algorithmic approach", "MatTransposeMatMult", algTypes, nalg, algTypes[alg], &alg, &flg));
2416:     PetscOptionsEnd();
2417:   }
2418:   if (flg) PetscCall(MatProductSetAlgorithm(C, algTypes[alg]));

2420:   C->ops->productsymbolic = MatProductSymbolic_AtB_MPIAIJ_MPIAIJ;
2421:   PetscFunctionReturn(PETSC_SUCCESS);
2422: }

2424: static PetscErrorCode MatProductSetFromOptions_MPIAIJ_PtAP(Mat C)
2425: {
2426:   Mat_Product *product = C->product;
2427:   Mat          A = product->A, P = product->B;
2428:   MPI_Comm     comm;
2429:   PetscBool    flg;
2430:   PetscInt     alg = 1; /* set default algorithm */
2431: #if !PetscDefined(HAVE_HYPRE)
2432:   const char *algTypes[5] = {"scalable", "nonscalable", "allatonce", "allatonce_merged", "backend"};
2433:   PetscInt    nalg        = 5;
2434: #else
2435:   const char *algTypes[6] = {"scalable", "nonscalable", "allatonce", "allatonce_merged", "backend", "hypre"};
2436:   PetscInt    nalg        = 6;
2437: #endif
2438:   PetscInt pN = P->cmap->N;

2440:   PetscFunctionBegin;
2441:   /* Check matrix local sizes */
2442:   PetscCall(PetscObjectGetComm((PetscObject)C, &comm));
2443:   PetscCheck(A->rmap->rstart == P->rmap->rstart && A->rmap->rend == P->rmap->rend, PETSC_COMM_SELF, PETSC_ERR_ARG_SIZ, "Matrix local dimensions are incompatible, Arow (%" PetscInt_FMT ", %" PetscInt_FMT ") != Prow (%" PetscInt_FMT ",%" PetscInt_FMT ")",
2444:              A->rmap->rstart, A->rmap->rend, P->rmap->rstart, P->rmap->rend);
2445:   PetscCheck(A->cmap->rstart == P->rmap->rstart && A->cmap->rend == P->rmap->rend, PETSC_COMM_SELF, PETSC_ERR_ARG_SIZ, "Matrix local dimensions are incompatible, Acol (%" PetscInt_FMT ", %" PetscInt_FMT ") != Prow (%" PetscInt_FMT ",%" PetscInt_FMT ")",
2446:              A->cmap->rstart, A->cmap->rend, P->rmap->rstart, P->rmap->rend);

2448:   /* Set "nonscalable" as default algorithm */
2449:   PetscCall(PetscStrcmp(C->product->alg, "default", &flg));
2450:   if (flg) {
2451:     PetscCall(MatProductSetAlgorithm(C, algTypes[alg]));

2453:     /* Set "scalable" as default if BN and local nonzeros of A and B are large */
2454:     if (pN > 100000) {
2455:       MatInfo   Ainfo, Pinfo;
2456:       PetscInt  nz_local;
2457:       PetscBool alg_scalable = PETSC_FALSE;

2459:       PetscCall(MatGetInfo(A, MAT_LOCAL, &Ainfo));
2460:       PetscCall(MatGetInfo(P, MAT_LOCAL, &Pinfo));
2461:       nz_local = (PetscInt)(Ainfo.nz_allocated + Pinfo.nz_allocated);

2463:       if (pN > product->fill * nz_local) alg_scalable = PETSC_TRUE;
2464:       PetscCallMPI(MPIU_Allreduce(MPI_IN_PLACE, &alg_scalable, 1, MPI_C_BOOL, MPI_LOR, comm));

2466:       if (alg_scalable) {
2467:         alg = 0; /* scalable algorithm would 50% slower than nonscalable algorithm */
2468:         PetscCall(MatProductSetAlgorithm(C, algTypes[alg]));
2469:       }
2470:     }
2471:   }

2473:   /* Get runtime option */
2474:   if (product->api_user) {
2475:     PetscOptionsBegin(PetscObjectComm((PetscObject)C), ((PetscObject)C)->prefix, "MatPtAP", "Mat");
2476:     PetscCall(PetscOptionsEList("-matptap_via", "Algorithmic approach", "MatPtAP", algTypes, nalg, algTypes[alg], &alg, &flg));
2477:     PetscOptionsEnd();
2478:   } else {
2479:     PetscOptionsBegin(PetscObjectComm((PetscObject)C), ((PetscObject)C)->prefix, "MatProduct_PtAP", "Mat");
2480:     PetscCall(PetscOptionsEList("-mat_product_algorithm", "Algorithmic approach", "MatPtAP", algTypes, nalg, algTypes[alg], &alg, &flg));
2481:     PetscOptionsEnd();
2482:   }
2483:   if (flg) PetscCall(MatProductSetAlgorithm(C, algTypes[alg]));

2485:   C->ops->productsymbolic = MatProductSymbolic_PtAP_MPIAIJ_MPIAIJ;
2486:   PetscFunctionReturn(PETSC_SUCCESS);
2487: }

2489: static PetscErrorCode MatProductSetFromOptions_MPIAIJ_RARt(Mat C)
2490: {
2491:   Mat_Product *product = C->product;
2492:   Mat          A = product->A, R = product->B;

2494:   PetscFunctionBegin;
2495:   /* Check matrix local sizes */
2496:   PetscCheck(A->cmap->n == R->cmap->n && A->rmap->n == R->cmap->n, PETSC_COMM_SELF, PETSC_ERR_ARG_SIZ, "Matrix local dimensions are incompatible, A local (%" PetscInt_FMT ", %" PetscInt_FMT "), R local (%" PetscInt_FMT ",%" PetscInt_FMT ")", A->rmap->n,
2497:              A->rmap->n, R->rmap->n, R->cmap->n);

2499:   C->ops->productsymbolic = MatProductSymbolic_RARt_MPIAIJ_MPIAIJ;
2500:   PetscFunctionReturn(PETSC_SUCCESS);
2501: }

2503: /*
2504:  Set options for ABC = A*B*C = A*(B*C); ABC's algorithm must be chosen from AB's algorithm
2505: */
2506: static PetscErrorCode MatProductSetFromOptions_MPIAIJ_ABC(Mat C)
2507: {
2508:   Mat_Product *product     = C->product;
2509:   PetscBool    flg         = PETSC_FALSE;
2510:   PetscInt     alg         = 1; /* default algorithm */
2511:   const char  *algTypes[3] = {"scalable", "nonscalable", "seqmpi"};
2512:   PetscInt     nalg        = 3;

2514:   PetscFunctionBegin;
2515:   /* Set default algorithm */
2516:   PetscCall(PetscStrcmp(C->product->alg, "default", &flg));
2517:   if (flg) PetscCall(MatProductSetAlgorithm(C, algTypes[alg]));

2519:   /* Get runtime option */
2520:   if (product->api_user) {
2521:     PetscOptionsBegin(PetscObjectComm((PetscObject)C), ((PetscObject)C)->prefix, "MatMatMatMult", "Mat");
2522:     PetscCall(PetscOptionsEList("-matmatmatmult_via", "Algorithmic approach", "MatMatMatMult", algTypes, nalg, algTypes[alg], &alg, &flg));
2523:     PetscOptionsEnd();
2524:   } else {
2525:     PetscOptionsBegin(PetscObjectComm((PetscObject)C), ((PetscObject)C)->prefix, "MatProduct_ABC", "Mat");
2526:     PetscCall(PetscOptionsEList("-mat_product_algorithm", "Algorithmic approach", "MatProduct_ABC", algTypes, nalg, algTypes[alg], &alg, &flg));
2527:     PetscOptionsEnd();
2528:   }
2529:   if (flg) PetscCall(MatProductSetAlgorithm(C, algTypes[alg]));

2531:   C->ops->matmatmultsymbolic = MatMatMatMultSymbolic_MPIAIJ_MPIAIJ_MPIAIJ;
2532:   C->ops->productsymbolic    = MatProductSymbolic_ABC;
2533:   PetscFunctionReturn(PETSC_SUCCESS);
2534: }

2536: PETSC_INTERN PetscErrorCode MatProductSetFromOptions_MPIAIJ(Mat C)
2537: {
2538:   Mat_Product *product = C->product;

2540:   PetscFunctionBegin;
2541:   switch (product->type) {
2542:   case MATPRODUCT_AB:
2543:     PetscCall(MatProductSetFromOptions_MPIAIJ_AB(C));
2544:     break;
2545:   case MATPRODUCT_ABt:
2546:     PetscCall(MatProductSetFromOptions_MPIAIJ_ABt(C));
2547:     break;
2548:   case MATPRODUCT_AtB:
2549:     PetscCall(MatProductSetFromOptions_MPIAIJ_AtB(C));
2550:     break;
2551:   case MATPRODUCT_PtAP:
2552:     PetscCall(MatProductSetFromOptions_MPIAIJ_PtAP(C));
2553:     break;
2554:   case MATPRODUCT_RARt:
2555:     PetscCall(MatProductSetFromOptions_MPIAIJ_RARt(C));
2556:     break;
2557:   case MATPRODUCT_ABC:
2558:     PetscCall(MatProductSetFromOptions_MPIAIJ_ABC(C));
2559:     break;
2560:   default:
2561:     break;
2562:   }
2563:   PetscFunctionReturn(PETSC_SUCCESS);
2564: }