xref: /libCEED/rust/libceed-sys/c-src/backends/blocked/ceed-blocked-operator.c (revision 430758c82d46a36f0f94ae5e5cf52a5e5bc2b1e9)
14a2e7687Sjeremylt // Copyright (c) 2017-2018, Lawrence Livermore National Security, LLC.
24a2e7687Sjeremylt // Produced at the Lawrence Livermore National Laboratory. LLNL-CODE-734707.
34a2e7687Sjeremylt // All Rights reserved. See files LICENSE and NOTICE for details.
44a2e7687Sjeremylt //
54a2e7687Sjeremylt // This file is part of CEED, a collection of benchmarks, miniapps, software
64a2e7687Sjeremylt // libraries and APIs for efficient high-order finite element and spectral
74a2e7687Sjeremylt // element discretizations for exascale applications. For more information and
84a2e7687Sjeremylt // source code availability see http://github.com/ceed.
94a2e7687Sjeremylt //
104a2e7687Sjeremylt // The CEED research is supported by the Exascale Computing Project 17-SC-20-SC,
114a2e7687Sjeremylt // a collaborative effort of two U.S. Department of Energy organizations (Office
124a2e7687Sjeremylt // of Science and the National Nuclear Security Administration) responsible for
134a2e7687Sjeremylt // the planning and preparation of a capable exascale ecosystem, including
144a2e7687Sjeremylt // software, applications, hardware, advanced system engineering and early
154a2e7687Sjeremylt // testbed platforms, in support of the nation's exascale computing imperative.
164a2e7687Sjeremylt 
174a2e7687Sjeremylt #include "ceed-blocked.h"
184a2e7687Sjeremylt 
19f10650afSjeremylt //------------------------------------------------------------------------------
20f10650afSjeremylt // Setup Input/Output Fields
21f10650afSjeremylt //------------------------------------------------------------------------------
2289c6efa4Sjeremylt static int CeedOperatorSetupFields_Blocked(CeedQFunction qf,
2389c6efa4Sjeremylt     CeedOperator op, bool inOrOut,
244a2e7687Sjeremylt     CeedElemRestriction *blkrestr,
2591703d3fSjeremylt     CeedVector *fullevecs, CeedVector *evecs,
26aedaa0e5Sjeremylt     CeedVector *qvecs, CeedInt starte,
274a2e7687Sjeremylt     CeedInt numfields, CeedInt Q) {
284d537eeaSYohann   CeedInt dim, ierr, ncomp, size, P;
29aedaa0e5Sjeremylt   Ceed ceed;
30aedaa0e5Sjeremylt   ierr = CeedOperatorGetCeed(op, &ceed); CeedChk(ierr);
31d1bcdac9Sjeremylt   CeedBasis basis;
32d1bcdac9Sjeremylt   CeedElemRestriction r;
33aedaa0e5Sjeremylt   CeedOperatorField *opfields;
34aedaa0e5Sjeremylt   CeedQFunctionField *qffields;
35fe2413ffSjeremylt   if (inOrOut) {
36aedaa0e5Sjeremylt     ierr = CeedOperatorGetFields(op, NULL, &opfields);
37fe2413ffSjeremylt     CeedChk(ierr);
38aedaa0e5Sjeremylt     ierr = CeedQFunctionGetFields(qf, NULL, &qffields);
39fe2413ffSjeremylt     CeedChk(ierr);
40fe2413ffSjeremylt   } else {
41aedaa0e5Sjeremylt     ierr = CeedOperatorGetFields(op, &opfields, NULL);
42fe2413ffSjeremylt     CeedChk(ierr);
43aedaa0e5Sjeremylt     ierr = CeedQFunctionGetFields(qf, &qffields, NULL);
44fe2413ffSjeremylt     CeedChk(ierr);
45fe2413ffSjeremylt   }
464a2e7687Sjeremylt   const CeedInt blksize = 8;
474a2e7687Sjeremylt 
484a2e7687Sjeremylt   // Loop over fields
494a2e7687Sjeremylt   for (CeedInt i=0; i<numfields; i++) {
50d1bcdac9Sjeremylt     CeedEvalMode emode;
51aedaa0e5Sjeremylt     ierr = CeedQFunctionFieldGetEvalMode(qffields[i], &emode); CeedChk(ierr);
524a2e7687Sjeremylt 
534a2e7687Sjeremylt     if (emode != CEED_EVAL_WEIGHT) {
54aedaa0e5Sjeremylt       ierr = CeedOperatorFieldGetElemRestriction(opfields[i], &r);
55d1bcdac9Sjeremylt       CeedChk(ierr);
564ce2993fSjeremylt       ierr = CeedElemRestrictionGetCeed(r, &ceed); CeedChk(ierr);
57d979a051Sjeremylt       CeedInt nelem, elemsize, lsize, compstride;
584ce2993fSjeremylt       ierr = CeedElemRestrictionGetNumElements(r, &nelem); CeedChk(ierr);
594ce2993fSjeremylt       ierr = CeedElemRestrictionGetElementSize(r, &elemsize); CeedChk(ierr);
60d979a051Sjeremylt       ierr = CeedElemRestrictionGetLVectorSize(r, &lsize); CeedChk(ierr);
614ce2993fSjeremylt       ierr = CeedElemRestrictionGetNumComponents(r, &ncomp); CeedChk(ierr);
62bd33150aSjeremylt 
63bd33150aSjeremylt       const CeedInt *offsets = NULL;
64bd33150aSjeremylt       ierr = CeedElemRestrictionGetOffsets(r, CEED_MEM_HOST, &offsets);
65bd33150aSjeremylt       CeedChk(ierr);
66bd33150aSjeremylt       if (offsets) {
67d979a051Sjeremylt         ierr = CeedElemRestrictionGetCompStride(r, &compstride); CeedChk(ierr);
68d979a051Sjeremylt         ierr = CeedElemRestrictionCreateBlocked(ceed, nelem, elemsize,
69d979a051Sjeremylt                                                 blksize, ncomp, compstride,
70d979a051Sjeremylt                                                 lsize, CEED_MEM_HOST,
71bd33150aSjeremylt                                                 CEED_COPY_VALUES, offsets,
727509a596Sjeremylt                                                 &blkrestr[i+starte]);
734a2e7687Sjeremylt         CeedChk(ierr);
747509a596Sjeremylt       } else {
757509a596Sjeremylt         CeedInt strides[3];
767509a596Sjeremylt         ierr = CeedElemRestrictionGetStrides(r, &strides); CeedChk(ierr);
777509a596Sjeremylt         ierr = CeedElemRestrictionCreateBlockedStrided(ceed, nelem, elemsize,
78d979a051Sjeremylt                blksize, ncomp, lsize, strides, &blkrestr[i+starte]);
797509a596Sjeremylt         CeedChk(ierr);
807509a596Sjeremylt       }
81*430758c8SJeremy L Thompson       ierr = CeedElemRestrictionRestoreOffsets(r, &offsets); CeedChk(ierr);
82aedaa0e5Sjeremylt       ierr = CeedElemRestrictionCreateVector(blkrestr[i+starte], NULL,
8391703d3fSjeremylt                                              &fullevecs[i+starte]);
844a2e7687Sjeremylt       CeedChk(ierr);
854a2e7687Sjeremylt     }
864a2e7687Sjeremylt 
874a2e7687Sjeremylt     switch(emode) {
884a2e7687Sjeremylt     case CEED_EVAL_NONE:
894d537eeaSYohann       ierr = CeedQFunctionFieldGetSize(qffields[i], &size); CeedChk(ierr);
904d537eeaSYohann       ierr = CeedVectorCreate(ceed, Q*size*blksize, &qvecs[i]); CeedChk(ierr);
91aedaa0e5Sjeremylt       break;
92aedaa0e5Sjeremylt     case CEED_EVAL_INTERP:
934d537eeaSYohann       ierr = CeedQFunctionFieldGetSize(qffields[i], &size); CeedChk(ierr);
944d1cd9fcSJeremy L Thompson       ierr = CeedElemRestrictionGetElementSize(r, &P);
954d1cd9fcSJeremy L Thompson       CeedChk(ierr);
964d537eeaSYohann       ierr = CeedVectorCreate(ceed, P*size*blksize, &evecs[i]); CeedChk(ierr);
974d537eeaSYohann       ierr = CeedVectorCreate(ceed, Q*size*blksize, &qvecs[i]); CeedChk(ierr);
984a2e7687Sjeremylt       break;
994a2e7687Sjeremylt     case CEED_EVAL_GRAD:
100aedaa0e5Sjeremylt       ierr = CeedOperatorFieldGetBasis(opfields[i], &basis); CeedChk(ierr);
1014d537eeaSYohann       ierr = CeedQFunctionFieldGetSize(qffields[i], &size); CeedChk(ierr);
102d1bcdac9Sjeremylt       ierr = CeedBasisGetDimension(basis, &dim); CeedChk(ierr);
1034d1cd9fcSJeremy L Thompson       ierr = CeedElemRestrictionGetElementSize(r, &P);
1044d1cd9fcSJeremy L Thompson       CeedChk(ierr);
1054d537eeaSYohann       ierr = CeedVectorCreate(ceed, P*size/dim*blksize, &evecs[i]); CeedChk(ierr);
1064d537eeaSYohann       ierr = CeedVectorCreate(ceed, Q*size*blksize, &qvecs[i]); CeedChk(ierr);
1074a2e7687Sjeremylt       break;
1084a2e7687Sjeremylt     case CEED_EVAL_WEIGHT: // Only on input fields
109aedaa0e5Sjeremylt       ierr = CeedOperatorFieldGetBasis(opfields[i], &basis); CeedChk(ierr);
110aedaa0e5Sjeremylt       ierr = CeedVectorCreate(ceed, Q*blksize, &qvecs[i]); CeedChk(ierr);
111d1bcdac9Sjeremylt       ierr = CeedBasisApply(basis, blksize, CEED_NOTRANSPOSE,
112a7b7f929Sjeremylt                             CEED_EVAL_WEIGHT, CEED_VECTOR_NONE, qvecs[i]);
113a7b7f929Sjeremylt       CeedChk(ierr);
114aedaa0e5Sjeremylt 
1154a2e7687Sjeremylt       break;
1164a2e7687Sjeremylt     case CEED_EVAL_DIV:
1174d537eeaSYohann       break; // Not implemented
1184a2e7687Sjeremylt     case CEED_EVAL_CURL:
1194d537eeaSYohann       break; // Not implemented
1204a2e7687Sjeremylt     }
1214a2e7687Sjeremylt   }
1224a2e7687Sjeremylt   return 0;
1234a2e7687Sjeremylt }
1244a2e7687Sjeremylt 
125f10650afSjeremylt //------------------------------------------------------------------------------
126f10650afSjeremylt // Setup Operator
127f10650afSjeremylt //------------------------------------------------------------------------------
1284a2e7687Sjeremylt static int CeedOperatorSetup_Blocked(CeedOperator op) {
1294a2e7687Sjeremylt   int ierr;
1304ce2993fSjeremylt   bool setupdone;
1314ce2993fSjeremylt   ierr = CeedOperatorGetSetupStatus(op, &setupdone); CeedChk(ierr);
1324ce2993fSjeremylt   if (setupdone) return 0;
133aedaa0e5Sjeremylt   Ceed ceed;
134aedaa0e5Sjeremylt   ierr = CeedOperatorGetCeed(op, &ceed); CeedChk(ierr);
1354ce2993fSjeremylt   CeedOperator_Blocked *impl;
1364ce2993fSjeremylt   ierr = CeedOperatorGetData(op, (void *)&impl); CeedChk(ierr);
1374ce2993fSjeremylt   CeedQFunction qf;
1384ce2993fSjeremylt   ierr = CeedOperatorGetQFunction(op, &qf); CeedChk(ierr);
1394ce2993fSjeremylt   CeedInt Q, numinputfields, numoutputfields;
1404ce2993fSjeremylt   ierr = CeedOperatorGetNumQuadraturePoints(op, &Q); CeedChk(ierr);
14116911fdaSjeremylt   ierr = CeedQFunctionGetIdentityStatus(qf, &impl->identityqf); CeedChk(ierr);
1424a2e7687Sjeremylt   ierr= CeedQFunctionGetNumArgs(qf, &numinputfields, &numoutputfields);
1434a2e7687Sjeremylt   CeedChk(ierr);
144d1bcdac9Sjeremylt   CeedOperatorField *opinputfields, *opoutputfields;
145d1bcdac9Sjeremylt   ierr = CeedOperatorGetFields(op, &opinputfields, &opoutputfields);
146d1bcdac9Sjeremylt   CeedChk(ierr);
147d1bcdac9Sjeremylt   CeedQFunctionField *qfinputfields, *qfoutputfields;
148d1bcdac9Sjeremylt   ierr = CeedQFunctionGetFields(qf, &qfinputfields, &qfoutputfields);
149d1bcdac9Sjeremylt   CeedChk(ierr);
1504a2e7687Sjeremylt 
1514a2e7687Sjeremylt   // Allocate
152aedaa0e5Sjeremylt   ierr = CeedCalloc(numinputfields + numoutputfields, &impl->blkrestr);
1534a2e7687Sjeremylt   CeedChk(ierr);
154aedaa0e5Sjeremylt   ierr = CeedCalloc(numinputfields + numoutputfields, &impl->evecs);
1554a2e7687Sjeremylt   CeedChk(ierr);
156aedaa0e5Sjeremylt   ierr = CeedCalloc(numinputfields + numoutputfields, &impl->edata);
1574a2e7687Sjeremylt   CeedChk(ierr);
1584a2e7687Sjeremylt 
15916c359e6Sjeremylt   ierr = CeedCalloc(16, &impl->inputstate); CeedChk(ierr);
16091703d3fSjeremylt   ierr = CeedCalloc(16, &impl->evecsin); CeedChk(ierr);
16191703d3fSjeremylt   ierr = CeedCalloc(16, &impl->evecsout); CeedChk(ierr);
162aedaa0e5Sjeremylt   ierr = CeedCalloc(16, &impl->qvecsin); CeedChk(ierr);
163aedaa0e5Sjeremylt   ierr = CeedCalloc(16, &impl->qvecsout); CeedChk(ierr);
1644a2e7687Sjeremylt 
165aedaa0e5Sjeremylt   impl->numein = numinputfields; impl->numeout = numoutputfields;
166aedaa0e5Sjeremylt 
1674a2e7687Sjeremylt   // Set up infield and outfield pointer arrays
1684a2e7687Sjeremylt   // Infields
169aedaa0e5Sjeremylt   ierr = CeedOperatorSetupFields_Blocked(qf, op, 0, impl->blkrestr,
17091703d3fSjeremylt                                          impl->evecs, impl->evecsin,
17191703d3fSjeremylt                                          impl->qvecsin, 0,
172aedaa0e5Sjeremylt                                          numinputfields, Q);
1734a2e7687Sjeremylt   CeedChk(ierr);
1744a2e7687Sjeremylt   // Outfields
175aedaa0e5Sjeremylt   ierr = CeedOperatorSetupFields_Blocked(qf, op, 1, impl->blkrestr,
17691703d3fSjeremylt                                          impl->evecs, impl->evecsout,
17791703d3fSjeremylt                                          impl->qvecsout, numinputfields,
17891703d3fSjeremylt                                          numoutputfields, Q);
1794a2e7687Sjeremylt   CeedChk(ierr);
180aedaa0e5Sjeremylt 
18116911fdaSjeremylt   // Identity QFunctions
18216911fdaSjeremylt   if (impl->identityqf) {
18316911fdaSjeremylt     CeedEvalMode inmode, outmode;
18416911fdaSjeremylt     CeedQFunctionField *infields, *outfields;
18516911fdaSjeremylt     ierr = CeedQFunctionGetFields(qf, &infields, &outfields); CeedChk(ierr);
18616911fdaSjeremylt 
18716911fdaSjeremylt     for (CeedInt i=0; i<numinputfields; i++) {
18816911fdaSjeremylt       ierr = CeedQFunctionFieldGetEvalMode(infields[i], &inmode);
18916911fdaSjeremylt       CeedChk(ierr);
19016911fdaSjeremylt       ierr = CeedQFunctionFieldGetEvalMode(outfields[i], &outmode);
19116911fdaSjeremylt       CeedChk(ierr);
19216911fdaSjeremylt 
19316911fdaSjeremylt       ierr = CeedVectorDestroy(&impl->qvecsout[i]); CeedChk(ierr);
19416911fdaSjeremylt       impl->qvecsout[i] = impl->qvecsin[i];
19555e4cc5bSjeremylt       ierr = CeedVectorAddReference(impl->qvecsin[i]); CeedChk(ierr);
19616911fdaSjeremylt     }
19716911fdaSjeremylt   }
19816911fdaSjeremylt 
1994ce2993fSjeremylt   ierr = CeedOperatorSetSetupDone(op); CeedChk(ierr);
2004a2e7687Sjeremylt 
2014a2e7687Sjeremylt   return 0;
2024a2e7687Sjeremylt }
2034a2e7687Sjeremylt 
204f10650afSjeremylt //------------------------------------------------------------------------------
205f10650afSjeremylt // Setup Operator Inputs
206f10650afSjeremylt //------------------------------------------------------------------------------
2071d102b48SJeremy L Thompson static inline int CeedOperatorSetupInputs_Blocked(CeedInt numinputfields,
2081d102b48SJeremy L Thompson     CeedQFunctionField *qfinputfields, CeedOperatorField *opinputfields,
2091d102b48SJeremy L Thompson     CeedVector invec, bool skipactive, CeedOperator_Blocked *impl,
21089c6efa4Sjeremylt     CeedRequest *request) {
2111d102b48SJeremy L Thompson   CeedInt ierr;
212d1bcdac9Sjeremylt   CeedEvalMode emode;
213d1bcdac9Sjeremylt   CeedVector vec;
21416c359e6Sjeremylt   uint64_t state;
2154a2e7687Sjeremylt 
2164a2e7687Sjeremylt   for (CeedInt i=0; i<numinputfields; i++) {
2171d102b48SJeremy L Thompson     // Get input vector
2181d102b48SJeremy L Thompson     ierr = CeedOperatorFieldGetVector(opinputfields[i], &vec); CeedChk(ierr);
2191d102b48SJeremy L Thompson     if (vec == CEED_VECTOR_ACTIVE) {
2201d102b48SJeremy L Thompson       if (skipactive)
2211d102b48SJeremy L Thompson         continue;
2221d102b48SJeremy L Thompson       else
2231d102b48SJeremy L Thompson         vec = invec;
2241d102b48SJeremy L Thompson     }
2251d102b48SJeremy L Thompson 
226d1bcdac9Sjeremylt     ierr = CeedQFunctionFieldGetEvalMode(qfinputfields[i], &emode);
227d1bcdac9Sjeremylt     CeedChk(ierr);
2284a2e7687Sjeremylt     if (emode == CEED_EVAL_WEIGHT) { // Skip
2294a2e7687Sjeremylt     } else {
2304a2e7687Sjeremylt       // Restrict
23116c359e6Sjeremylt       ierr = CeedVectorGetState(vec, &state); CeedChk(ierr);
23289c6efa4Sjeremylt       if (state != impl->inputstate[i] || vec == invec) {
2334a2e7687Sjeremylt         ierr = CeedElemRestrictionApply(impl->blkrestr[i], CEED_NOTRANSPOSE,
234a8d32208Sjeremylt                                         vec, impl->evecs[i], request);
235a8d32208Sjeremylt         CeedChk(ierr);
23616c359e6Sjeremylt         impl->inputstate[i] = state;
23716c359e6Sjeremylt       }
2384a2e7687Sjeremylt       // Get evec
2394a2e7687Sjeremylt       ierr = CeedVectorGetArrayRead(impl->evecs[i], CEED_MEM_HOST,
2404a2e7687Sjeremylt                                     (const CeedScalar **) &impl->edata[i]);
2414a2e7687Sjeremylt       CeedChk(ierr);
2424a2e7687Sjeremylt     }
2434a2e7687Sjeremylt   }
2441d102b48SJeremy L Thompson   return 0;
2454a2e7687Sjeremylt }
2464a2e7687Sjeremylt 
247f10650afSjeremylt //------------------------------------------------------------------------------
248f10650afSjeremylt // Input Basis Action
249f10650afSjeremylt //------------------------------------------------------------------------------
2501d102b48SJeremy L Thompson static inline int CeedOperatorInputBasis_Blocked(CeedInt e, CeedInt Q,
2511d102b48SJeremy L Thompson     CeedQFunctionField *qfinputfields, CeedOperatorField *opinputfields,
2521d102b48SJeremy L Thompson     CeedInt numinputfields, CeedInt blksize, bool skipactive,
2531d102b48SJeremy L Thompson     CeedOperator_Blocked *impl) {
2541d102b48SJeremy L Thompson   CeedInt ierr;
2551d102b48SJeremy L Thompson   CeedInt dim, elemsize, size;
2561d102b48SJeremy L Thompson   CeedElemRestriction Erestrict;
2571d102b48SJeremy L Thompson   CeedEvalMode emode;
2581d102b48SJeremy L Thompson   CeedBasis basis;
2591d102b48SJeremy L Thompson 
2604a2e7687Sjeremylt   for (CeedInt i=0; i<numinputfields; i++) {
2611d102b48SJeremy L Thompson     // Skip active input
2621d102b48SJeremy L Thompson     if (skipactive) {
2631d102b48SJeremy L Thompson       CeedVector vec;
2641d102b48SJeremy L Thompson       ierr = CeedOperatorFieldGetVector(opinputfields[i], &vec); CeedChk(ierr);
2651d102b48SJeremy L Thompson       if (vec == CEED_VECTOR_ACTIVE)
2661d102b48SJeremy L Thompson         continue;
2671d102b48SJeremy L Thompson     }
2681d102b48SJeremy L Thompson 
2694d537eeaSYohann     // Get elemsize, emode, size
270d1bcdac9Sjeremylt     ierr = CeedOperatorFieldGetElemRestriction(opinputfields[i], &Erestrict);
271d1bcdac9Sjeremylt     CeedChk(ierr);
272d1bcdac9Sjeremylt     ierr = CeedElemRestrictionGetElementSize(Erestrict, &elemsize);
273d1bcdac9Sjeremylt     CeedChk(ierr);
274d1bcdac9Sjeremylt     ierr = CeedQFunctionFieldGetEvalMode(qfinputfields[i], &emode);
275d1bcdac9Sjeremylt     CeedChk(ierr);
2764d537eeaSYohann     ierr = CeedQFunctionFieldGetSize(qfinputfields[i], &size); CeedChk(ierr);
2774a2e7687Sjeremylt     // Basis action
2784a2e7687Sjeremylt     switch(emode) {
2794a2e7687Sjeremylt     case CEED_EVAL_NONE:
280aedaa0e5Sjeremylt       ierr = CeedVectorSetArray(impl->qvecsin[i], CEED_MEM_HOST,
281aedaa0e5Sjeremylt                                 CEED_USE_POINTER,
2824d537eeaSYohann                                 &impl->edata[i][e*Q*size]); CeedChk(ierr);
2834a2e7687Sjeremylt       break;
2844a2e7687Sjeremylt     case CEED_EVAL_INTERP:
28589c6efa4Sjeremylt       ierr = CeedOperatorFieldGetBasis(opinputfields[i], &basis); CeedChk(ierr);
28691703d3fSjeremylt       ierr = CeedVectorSetArray(impl->evecsin[i], CEED_MEM_HOST,
287aedaa0e5Sjeremylt                                 CEED_USE_POINTER,
2884d537eeaSYohann                                 &impl->edata[i][e*elemsize*size]);
2894a2e7687Sjeremylt       CeedChk(ierr);
290d1bcdac9Sjeremylt       ierr = CeedBasisApply(basis, blksize, CEED_NOTRANSPOSE,
29191703d3fSjeremylt                             CEED_EVAL_INTERP, impl->evecsin[i],
292aedaa0e5Sjeremylt                             impl->qvecsin[i]); CeedChk(ierr);
2934a2e7687Sjeremylt       break;
2944a2e7687Sjeremylt     case CEED_EVAL_GRAD:
29589c6efa4Sjeremylt       ierr = CeedOperatorFieldGetBasis(opinputfields[i], &basis); CeedChk(ierr);
2964d537eeaSYohann       ierr = CeedBasisGetDimension(basis, &dim); CeedChk(ierr);
29791703d3fSjeremylt       ierr = CeedVectorSetArray(impl->evecsin[i], CEED_MEM_HOST,
298aedaa0e5Sjeremylt                                 CEED_USE_POINTER,
2994d537eeaSYohann                                 &impl->edata[i][e*elemsize*size/dim]);
3004a2e7687Sjeremylt       CeedChk(ierr);
301d1bcdac9Sjeremylt       ierr = CeedBasisApply(basis, blksize, CEED_NOTRANSPOSE,
30291703d3fSjeremylt                             CEED_EVAL_GRAD, impl->evecsin[i],
303aedaa0e5Sjeremylt                             impl->qvecsin[i]); CeedChk(ierr);
3044a2e7687Sjeremylt       break;
3054a2e7687Sjeremylt     case CEED_EVAL_WEIGHT:
3064a2e7687Sjeremylt       break;  // No action
307bbfacfcdSjeremylt     // LCOV_EXCL_START
3084a2e7687Sjeremylt     case CEED_EVAL_DIV:
3091d102b48SJeremy L Thompson     case CEED_EVAL_CURL: {
3101d102b48SJeremy L Thompson       ierr = CeedOperatorFieldGetBasis(opinputfields[i], &basis);
3111d102b48SJeremy L Thompson       CeedChk(ierr);
3121d102b48SJeremy L Thompson       Ceed ceed;
3131d102b48SJeremy L Thompson       ierr = CeedBasisGetCeed(basis, &ceed); CeedChk(ierr);
3141d102b48SJeremy L Thompson       return CeedError(ceed, 1, "Ceed evaluation mode not implemented");
315bbfacfcdSjeremylt       // LCOV_EXCL_STOP
3164a2e7687Sjeremylt     }
3174a2e7687Sjeremylt     }
31889c6efa4Sjeremylt   }
3191d102b48SJeremy L Thompson   return 0;
32089c6efa4Sjeremylt }
3214a2e7687Sjeremylt 
322f10650afSjeremylt //------------------------------------------------------------------------------
323f10650afSjeremylt // Output Basis Action
324f10650afSjeremylt //------------------------------------------------------------------------------
3251d102b48SJeremy L Thompson static inline int CeedOperatorOutputBasis_Blocked(CeedInt e, CeedInt Q,
3261d102b48SJeremy L Thompson     CeedQFunctionField *qfoutputfields, CeedOperatorField *opoutputfields,
3271d102b48SJeremy L Thompson     CeedInt blksize, CeedInt numinputfields, CeedInt numoutputfields,
3281d102b48SJeremy L Thompson     CeedOperator op, CeedOperator_Blocked *impl) {
3291d102b48SJeremy L Thompson   CeedInt ierr;
3301d102b48SJeremy L Thompson   CeedInt dim, elemsize, size;
3311d102b48SJeremy L Thompson   CeedElemRestriction Erestrict;
3321d102b48SJeremy L Thompson   CeedEvalMode emode;
3331d102b48SJeremy L Thompson   CeedBasis basis;
3341d102b48SJeremy L Thompson 
3354a2e7687Sjeremylt   for (CeedInt i=0; i<numoutputfields; i++) {
3364d537eeaSYohann     // Get elemsize, emode, size
337d1bcdac9Sjeremylt     ierr = CeedOperatorFieldGetElemRestriction(opoutputfields[i], &Erestrict);
338d1bcdac9Sjeremylt     CeedChk(ierr);
33989c6efa4Sjeremylt     ierr = CeedElemRestrictionGetElementSize(Erestrict, &elemsize);
34089c6efa4Sjeremylt     CeedChk(ierr);
341d1bcdac9Sjeremylt     ierr = CeedQFunctionFieldGetEvalMode(qfoutputfields[i], &emode);
342d1bcdac9Sjeremylt     CeedChk(ierr);
3434d537eeaSYohann     ierr = CeedQFunctionFieldGetSize(qfoutputfields[i], &size); CeedChk(ierr);
3444a2e7687Sjeremylt     // Basis action
3454a2e7687Sjeremylt     switch(emode) {
3464a2e7687Sjeremylt     case CEED_EVAL_NONE:
3474a2e7687Sjeremylt       break; // No action
3484a2e7687Sjeremylt     case CEED_EVAL_INTERP:
349d1bcdac9Sjeremylt       ierr = CeedOperatorFieldGetBasis(opoutputfields[i], &basis);
350d1bcdac9Sjeremylt       CeedChk(ierr);
35189c6efa4Sjeremylt       ierr = CeedVectorSetArray(impl->evecsout[i], CEED_MEM_HOST,
35289c6efa4Sjeremylt                                 CEED_USE_POINTER,
3534d537eeaSYohann                                 &impl->edata[i + numinputfields][e*elemsize*size]);
35489c6efa4Sjeremylt       CeedChk(ierr);
355aedaa0e5Sjeremylt       ierr = CeedBasisApply(basis, blksize, CEED_TRANSPOSE,
356aedaa0e5Sjeremylt                             CEED_EVAL_INTERP, impl->qvecsout[i],
35791703d3fSjeremylt                             impl->evecsout[i]); CeedChk(ierr);
3584a2e7687Sjeremylt       break;
3594a2e7687Sjeremylt     case CEED_EVAL_GRAD:
360d1bcdac9Sjeremylt       ierr = CeedOperatorFieldGetBasis(opoutputfields[i], &basis);
361d1bcdac9Sjeremylt       CeedChk(ierr);
3624d537eeaSYohann       ierr = CeedBasisGetDimension(basis, &dim); CeedChk(ierr);
36389c6efa4Sjeremylt       ierr = CeedVectorSetArray(impl->evecsout[i], CEED_MEM_HOST,
36489c6efa4Sjeremylt                                 CEED_USE_POINTER,
3654d537eeaSYohann                                 &impl->edata[i + numinputfields][e*elemsize*size/dim]);
36689c6efa4Sjeremylt       CeedChk(ierr);
367d1bcdac9Sjeremylt       ierr = CeedBasisApply(basis, blksize, CEED_TRANSPOSE,
368aedaa0e5Sjeremylt                             CEED_EVAL_GRAD, impl->qvecsout[i],
36991703d3fSjeremylt                             impl->evecsout[i]); CeedChk(ierr);
3704a2e7687Sjeremylt       break;
371c042f62fSJeremy L Thompson     // LCOV_EXCL_START
372bbfacfcdSjeremylt     case CEED_EVAL_WEIGHT: {
3734ce2993fSjeremylt       Ceed ceed;
3744ce2993fSjeremylt       ierr = CeedOperatorGetCeed(op, &ceed); CeedChk(ierr);
3751d102b48SJeremy L Thompson       return CeedError(ceed, 1, "CEED_EVAL_WEIGHT cannot be an output "
3761d102b48SJeremy L Thompson                        "evaluation mode");
3774ce2993fSjeremylt     }
3784a2e7687Sjeremylt     case CEED_EVAL_DIV:
3791d102b48SJeremy L Thompson     case CEED_EVAL_CURL: {
3801d102b48SJeremy L Thompson       Ceed ceed;
3811d102b48SJeremy L Thompson       ierr = CeedOperatorGetCeed(op, &ceed); CeedChk(ierr);
3821d102b48SJeremy L Thompson       return CeedError(ceed, 1, "Ceed evaluation mode not implemented");
383bbfacfcdSjeremylt       // LCOV_EXCL_STOP
3844a2e7687Sjeremylt     }
38589c6efa4Sjeremylt     }
38689c6efa4Sjeremylt   }
3871d102b48SJeremy L Thompson   return 0;
3881d102b48SJeremy L Thompson }
3891d102b48SJeremy L Thompson 
390f10650afSjeremylt //------------------------------------------------------------------------------
391f10650afSjeremylt // Restore Input Vectors
392f10650afSjeremylt //------------------------------------------------------------------------------
3931d102b48SJeremy L Thompson static inline int CeedOperatorRestoreInputs_Blocked(CeedInt numinputfields,
3941d102b48SJeremy L Thompson     CeedQFunctionField *qfinputfields, CeedOperatorField *opinputfields,
3951d102b48SJeremy L Thompson     bool skipactive, CeedOperator_Blocked *impl) {
3961d102b48SJeremy L Thompson   CeedInt ierr;
3971d102b48SJeremy L Thompson   CeedEvalMode emode;
3981d102b48SJeremy L Thompson 
3991d102b48SJeremy L Thompson   for (CeedInt i=0; i<numinputfields; i++) {
4001d102b48SJeremy L Thompson     // Skip active inputs
4011d102b48SJeremy L Thompson     if (skipactive) {
4021d102b48SJeremy L Thompson       CeedVector vec;
4031d102b48SJeremy L Thompson       ierr = CeedOperatorFieldGetVector(opinputfields[i], &vec); CeedChk(ierr);
4041d102b48SJeremy L Thompson       if (vec == CEED_VECTOR_ACTIVE)
4051d102b48SJeremy L Thompson         continue;
4061d102b48SJeremy L Thompson     }
4071d102b48SJeremy L Thompson     ierr = CeedQFunctionFieldGetEvalMode(qfinputfields[i], &emode);
4081d102b48SJeremy L Thompson     CeedChk(ierr);
4091d102b48SJeremy L Thompson     if (emode == CEED_EVAL_WEIGHT) { // Skip
4101d102b48SJeremy L Thompson     } else {
4111d102b48SJeremy L Thompson       ierr = CeedVectorRestoreArrayRead(impl->evecs[i],
4121d102b48SJeremy L Thompson                                         (const CeedScalar **) &impl->edata[i]);
4131d102b48SJeremy L Thompson       CeedChk(ierr);
4141d102b48SJeremy L Thompson     }
4151d102b48SJeremy L Thompson   }
4161d102b48SJeremy L Thompson   return 0;
4171d102b48SJeremy L Thompson }
4181d102b48SJeremy L Thompson 
419f10650afSjeremylt //------------------------------------------------------------------------------
420f10650afSjeremylt // Operator Apply
421f10650afSjeremylt //------------------------------------------------------------------------------
4221d102b48SJeremy L Thompson static int CeedOperatorApply_Blocked(CeedOperator op, CeedVector invec,
4231d102b48SJeremy L Thompson                                      CeedVector outvec,
4241d102b48SJeremy L Thompson                                      CeedRequest *request) {
4251d102b48SJeremy L Thompson   int ierr;
4261d102b48SJeremy L Thompson   CeedOperator_Blocked *impl;
4271d102b48SJeremy L Thompson   ierr = CeedOperatorGetData(op, (void *)&impl); CeedChk(ierr);
4281d102b48SJeremy L Thompson   const CeedInt blksize = 8;
4291d102b48SJeremy L Thompson   CeedInt Q, numinputfields, numoutputfields, numelements, size;
4301d102b48SJeremy L Thompson   ierr = CeedOperatorGetNumElements(op, &numelements); CeedChk(ierr);
4311d102b48SJeremy L Thompson   ierr = CeedOperatorGetNumQuadraturePoints(op, &Q); CeedChk(ierr);
4321d102b48SJeremy L Thompson   CeedInt nblks = (numelements/blksize) + !!(numelements%blksize);
4331d102b48SJeremy L Thompson   CeedQFunction qf;
4341d102b48SJeremy L Thompson   ierr = CeedOperatorGetQFunction(op, &qf); CeedChk(ierr);
4351d102b48SJeremy L Thompson   ierr= CeedQFunctionGetNumArgs(qf, &numinputfields, &numoutputfields);
4361d102b48SJeremy L Thompson   CeedChk(ierr);
4371d102b48SJeremy L Thompson   CeedOperatorField *opinputfields, *opoutputfields;
4381d102b48SJeremy L Thompson   ierr = CeedOperatorGetFields(op, &opinputfields, &opoutputfields);
4391d102b48SJeremy L Thompson   CeedChk(ierr);
4401d102b48SJeremy L Thompson   CeedQFunctionField *qfinputfields, *qfoutputfields;
4411d102b48SJeremy L Thompson   ierr = CeedQFunctionGetFields(qf, &qfinputfields, &qfoutputfields);
4421d102b48SJeremy L Thompson   CeedChk(ierr);
4431d102b48SJeremy L Thompson   CeedEvalMode emode;
4441d102b48SJeremy L Thompson   CeedVector vec;
4451d102b48SJeremy L Thompson 
4461d102b48SJeremy L Thompson   // Setup
4471d102b48SJeremy L Thompson   ierr = CeedOperatorSetup_Blocked(op); CeedChk(ierr);
4481d102b48SJeremy L Thompson 
4491d102b48SJeremy L Thompson   // Input Evecs and Restriction
4501d102b48SJeremy L Thompson   ierr = CeedOperatorSetupInputs_Blocked(numinputfields, qfinputfields,
4517f823360Sjeremylt                                          opinputfields, invec, false, impl,
4527f823360Sjeremylt                                          request); CeedChk(ierr);
4531d102b48SJeremy L Thompson 
4541d102b48SJeremy L Thompson   // Output Evecs
4551d102b48SJeremy L Thompson   for (CeedInt i=0; i<numoutputfields; i++) {
4561d102b48SJeremy L Thompson     ierr = CeedVectorGetArray(impl->evecs[i+impl->numein], CEED_MEM_HOST,
4571d102b48SJeremy L Thompson                               &impl->edata[i + numinputfields]); CeedChk(ierr);
4581d102b48SJeremy L Thompson   }
4591d102b48SJeremy L Thompson 
4601d102b48SJeremy L Thompson   // Loop through elements
4611d102b48SJeremy L Thompson   for (CeedInt e=0; e<nblks*blksize; e+=blksize) {
4621d102b48SJeremy L Thompson     // Output pointers
4631d102b48SJeremy L Thompson     for (CeedInt i=0; i<numoutputfields; i++) {
4641d102b48SJeremy L Thompson       ierr = CeedQFunctionFieldGetEvalMode(qfoutputfields[i], &emode);
4651d102b48SJeremy L Thompson       CeedChk(ierr);
4661d102b48SJeremy L Thompson       if (emode == CEED_EVAL_NONE) {
4671d102b48SJeremy L Thompson         ierr = CeedQFunctionFieldGetSize(qfoutputfields[i], &size);
4681d102b48SJeremy L Thompson         CeedChk(ierr);
4691d102b48SJeremy L Thompson         ierr = CeedVectorSetArray(impl->qvecsout[i], CEED_MEM_HOST,
4701d102b48SJeremy L Thompson                                   CEED_USE_POINTER,
4711d102b48SJeremy L Thompson                                   &impl->edata[i + numinputfields][e*Q*size]);
4721d102b48SJeremy L Thompson         CeedChk(ierr);
4731d102b48SJeremy L Thompson       }
4741d102b48SJeremy L Thompson     }
4751d102b48SJeremy L Thompson 
47616911fdaSjeremylt     // Input basis apply
47716911fdaSjeremylt     ierr = CeedOperatorInputBasis_Blocked(e, Q, qfinputfields, opinputfields,
47816911fdaSjeremylt                                           numinputfields, blksize, false, impl);
47916911fdaSjeremylt     CeedChk(ierr);
48016911fdaSjeremylt 
4811d102b48SJeremy L Thompson     // Q function
48216911fdaSjeremylt     if (!impl->identityqf) {
4831d102b48SJeremy L Thompson       ierr = CeedQFunctionApply(qf, Q*blksize, impl->qvecsin, impl->qvecsout);
4841d102b48SJeremy L Thompson       CeedChk(ierr);
48516911fdaSjeremylt     }
4861d102b48SJeremy L Thompson 
4871d102b48SJeremy L Thompson     // Output basis apply
4881d102b48SJeremy L Thompson     ierr = CeedOperatorOutputBasis_Blocked(e, Q, qfoutputfields, opoutputfields,
4897f823360Sjeremylt                                            blksize, numinputfields,
4907f823360Sjeremylt                                            numoutputfields, op, impl);
4911d102b48SJeremy L Thompson     CeedChk(ierr);
4921d102b48SJeremy L Thompson   }
49389c6efa4Sjeremylt 
49489c6efa4Sjeremylt   // Output restriction
49589c6efa4Sjeremylt   for (CeedInt i=0; i<numoutputfields; i++) {
49689c6efa4Sjeremylt     // Restore evec
49789c6efa4Sjeremylt     ierr = CeedVectorRestoreArray(impl->evecs[i+impl->numein],
49889c6efa4Sjeremylt                                   &impl->edata[i + numinputfields]); CeedChk(ierr);
499d1bcdac9Sjeremylt     // Get output vector
500d1bcdac9Sjeremylt     ierr = CeedOperatorFieldGetVector(opoutputfields[i], &vec); CeedChk(ierr);
50189c6efa4Sjeremylt     // Active
502d1bcdac9Sjeremylt     if (vec == CEED_VECTOR_ACTIVE)
503d1bcdac9Sjeremylt       vec = outvec;
5044a2e7687Sjeremylt     // Restrict
505a8d32208Sjeremylt     ierr = CeedElemRestrictionApply(impl->blkrestr[i+impl->numein],
506a8d32208Sjeremylt                                     CEED_TRANSPOSE, impl->evecs[i+impl->numein],
507a8d32208Sjeremylt                                     vec, request); CeedChk(ierr);
50889c6efa4Sjeremylt 
5094a2e7687Sjeremylt   }
5104a2e7687Sjeremylt 
5114a2e7687Sjeremylt   // Restore input arrays
5121d102b48SJeremy L Thompson   ierr = CeedOperatorRestoreInputs_Blocked(numinputfields, qfinputfields,
5137f823360Sjeremylt          opinputfields, false, impl); CeedChk(ierr);
5141d102b48SJeremy L Thompson 
5151d102b48SJeremy L Thompson   return 0;
5161d102b48SJeremy L Thompson }
5171d102b48SJeremy L Thompson 
518f10650afSjeremylt //------------------------------------------------------------------------------
5191d102b48SJeremy L Thompson // Assemble Linear QFunction
520f10650afSjeremylt //------------------------------------------------------------------------------
5211d102b48SJeremy L Thompson static int CeedOperatorAssembleLinearQFunction_Blocked(CeedOperator op,
5221d102b48SJeremy L Thompson     CeedVector *assembled, CeedElemRestriction *rstr, CeedRequest *request) {
5231d102b48SJeremy L Thompson   int ierr;
5241d102b48SJeremy L Thompson   CeedOperator_Blocked *impl;
5251d102b48SJeremy L Thompson   ierr = CeedOperatorGetData(op, (void *)&impl); CeedChk(ierr);
5261d102b48SJeremy L Thompson   const CeedInt blksize = 8;
5271d102b48SJeremy L Thompson   CeedInt Q, numinputfields, numoutputfields, numelements, size;
5281d102b48SJeremy L Thompson   ierr = CeedOperatorGetNumElements(op, &numelements); CeedChk(ierr);
5291d102b48SJeremy L Thompson   ierr = CeedOperatorGetNumQuadraturePoints(op, &Q); CeedChk(ierr);
5301d102b48SJeremy L Thompson   CeedInt nblks = (numelements/blksize) + !!(numelements%blksize);
5311d102b48SJeremy L Thompson   CeedQFunction qf;
5321d102b48SJeremy L Thompson   ierr = CeedOperatorGetQFunction(op, &qf); CeedChk(ierr);
5331d102b48SJeremy L Thompson   ierr= CeedQFunctionGetNumArgs(qf, &numinputfields, &numoutputfields);
5341d102b48SJeremy L Thompson   CeedChk(ierr);
5351d102b48SJeremy L Thompson   CeedOperatorField *opinputfields, *opoutputfields;
5361d102b48SJeremy L Thompson   ierr = CeedOperatorGetFields(op, &opinputfields, &opoutputfields);
5371d102b48SJeremy L Thompson   CeedChk(ierr);
5381d102b48SJeremy L Thompson   CeedQFunctionField *qfinputfields, *qfoutputfields;
5391d102b48SJeremy L Thompson   ierr = CeedQFunctionGetFields(qf, &qfinputfields, &qfoutputfields);
5401d102b48SJeremy L Thompson   CeedChk(ierr);
5411d102b48SJeremy L Thompson   CeedVector vec, lvec;
5421d102b48SJeremy L Thompson   CeedInt numactivein = 0, numactiveout = 0;
54342ea3801Sjeremylt   CeedVector *activein = NULL;
5441d102b48SJeremy L Thompson   CeedScalar *a, *tmp;
5451d102b48SJeremy L Thompson   Ceed ceed;
5461d102b48SJeremy L Thompson   ierr = CeedOperatorGetCeed(op, &ceed); CeedChk(ierr);
5471d102b48SJeremy L Thompson 
5481d102b48SJeremy L Thompson   // Setup
5491d102b48SJeremy L Thompson   ierr = CeedOperatorSetup_Blocked(op); CeedChk(ierr);
5501d102b48SJeremy L Thompson 
55116911fdaSjeremylt   // Check for identity
55216911fdaSjeremylt   if (impl->identityqf)
55316911fdaSjeremylt     // LCOV_EXCL_START
55467db23e4Sjeremylt     return CeedError(ceed, 1, "Assembling identity qfunctions not supported");
55516911fdaSjeremylt   // LCOV_EXCL_STOP
55616911fdaSjeremylt 
5571d102b48SJeremy L Thompson   // Input Evecs and Restriction
5581d102b48SJeremy L Thompson   ierr = CeedOperatorSetupInputs_Blocked(numinputfields, qfinputfields,
5591d102b48SJeremy L Thompson                                          opinputfields, NULL, true, impl,
5601d102b48SJeremy L Thompson                                          request); CeedChk(ierr);
5611d102b48SJeremy L Thompson 
5621d102b48SJeremy L Thompson   // Count number of active input fields
5634a2e7687Sjeremylt   for (CeedInt i=0; i<numinputfields; i++) {
5641d102b48SJeremy L Thompson     // Get input vector
5651d102b48SJeremy L Thompson     ierr = CeedOperatorFieldGetVector(opinputfields[i], &vec); CeedChk(ierr);
5661d102b48SJeremy L Thompson     // Check if active input
5671d102b48SJeremy L Thompson     if (vec == CEED_VECTOR_ACTIVE) {
5681d102b48SJeremy L Thompson       ierr = CeedQFunctionFieldGetSize(qfinputfields[i], &size); CeedChk(ierr);
5691d102b48SJeremy L Thompson       ierr = CeedVectorSetValue(impl->qvecsin[i], 0.0); CeedChk(ierr);
5701d102b48SJeremy L Thompson       ierr = CeedVectorGetArray(impl->qvecsin[i], CEED_MEM_HOST, &tmp);
571d1bcdac9Sjeremylt       CeedChk(ierr);
5721d102b48SJeremy L Thompson       ierr = CeedRealloc(numactivein + size, &activein); CeedChk(ierr);
5731d102b48SJeremy L Thompson       for (CeedInt field=0; field<size; field++) {
57442ea3801Sjeremylt         ierr = CeedVectorCreate(ceed, Q*blksize, &activein[numactivein+field]);
57542ea3801Sjeremylt         CeedChk(ierr);
57642ea3801Sjeremylt         ierr = CeedVectorSetArray(activein[numactivein+field], CEED_MEM_HOST,
57742ea3801Sjeremylt                                   CEED_USE_POINTER, &tmp[field*Q*blksize]);
578112e3f70Sjeremylt         CeedChk(ierr);
5791d102b48SJeremy L Thompson       }
5801d102b48SJeremy L Thompson       numactivein += size;
5811d102b48SJeremy L Thompson       ierr = CeedVectorRestoreArray(impl->qvecsin[i], &tmp); CeedChk(ierr);
5821d102b48SJeremy L Thompson     }
5831d102b48SJeremy L Thompson   }
5841d102b48SJeremy L Thompson 
5851d102b48SJeremy L Thompson   // Count number of active output fields
5861d102b48SJeremy L Thompson   for (CeedInt i=0; i<numoutputfields; i++) {
5871d102b48SJeremy L Thompson     // Get output vector
5881d102b48SJeremy L Thompson     ierr = CeedOperatorFieldGetVector(opoutputfields[i], &vec); CeedChk(ierr);
5891d102b48SJeremy L Thompson     // Check if active output
5901d102b48SJeremy L Thompson     if (vec == CEED_VECTOR_ACTIVE) {
5911d102b48SJeremy L Thompson       ierr = CeedQFunctionFieldGetSize(qfoutputfields[i], &size); CeedChk(ierr);
5921d102b48SJeremy L Thompson       numactiveout += size;
5931d102b48SJeremy L Thompson     }
5941d102b48SJeremy L Thompson   }
5951d102b48SJeremy L Thompson 
5961d102b48SJeremy L Thompson   // Check sizes
5971d102b48SJeremy L Thompson   if (!numactivein || !numactiveout)
5981d102b48SJeremy L Thompson     // LCOV_EXCL_START
5991d102b48SJeremy L Thompson     return CeedError(ceed, 1, "Cannot assemble QFunction without active inputs "
6001d102b48SJeremy L Thompson                      "and outputs");
6011d102b48SJeremy L Thompson   // LCOV_EXCL_STOP
6021d102b48SJeremy L Thompson 
6031d102b48SJeremy L Thompson   // Setup lvec
6041d102b48SJeremy L Thompson   ierr = CeedVectorCreate(ceed, nblks*blksize*Q*numactivein*numactiveout,
6051d102b48SJeremy L Thompson                           &lvec); CeedChk(ierr);
6061d102b48SJeremy L Thompson   ierr = CeedVectorGetArray(lvec, CEED_MEM_HOST, &a); CeedChk(ierr);
6071d102b48SJeremy L Thompson 
6081d102b48SJeremy L Thompson   // Create output restriction
6097509a596Sjeremylt   CeedInt strides[3] = {1, Q, numactivein *numactiveout*Q};
610d979a051Sjeremylt   ierr = CeedElemRestrictionCreateStrided(ceed, numelements, Q,
611d979a051Sjeremylt                                           numactivein*numactiveout,
612d979a051Sjeremylt                                           numactivein*numactiveout*numelements*Q,
613d979a051Sjeremylt                                           strides, rstr); CeedChk(ierr);
6141d102b48SJeremy L Thompson   // Create assembled vector
6151d102b48SJeremy L Thompson   ierr = CeedVectorCreate(ceed, numelements*Q*numactivein*numactiveout,
6161d102b48SJeremy L Thompson                           assembled); CeedChk(ierr);
6171d102b48SJeremy L Thompson 
6181d102b48SJeremy L Thompson   // Loop through elements
6191d102b48SJeremy L Thompson   for (CeedInt e=0; e<nblks*blksize; e+=blksize) {
6201d102b48SJeremy L Thompson     // Input basis apply
6211d102b48SJeremy L Thompson     ierr = CeedOperatorInputBasis_Blocked(e, Q, qfinputfields, opinputfields,
6221d102b48SJeremy L Thompson                                           numinputfields, blksize, true, impl);
6231d102b48SJeremy L Thompson     CeedChk(ierr);
6241d102b48SJeremy L Thompson 
6251d102b48SJeremy L Thompson     // Assemble QFunction
6261d102b48SJeremy L Thompson     for (CeedInt in=0; in<numactivein; in++) {
6271d102b48SJeremy L Thompson       // Set Inputs
62842ea3801Sjeremylt       ierr = CeedVectorSetValue(activein[in], 1.0); CeedChk(ierr);
62942ea3801Sjeremylt       if (numactivein > 1) {
63042ea3801Sjeremylt         ierr = CeedVectorSetValue(activein[(in+numactivein-1)%numactivein],
63142ea3801Sjeremylt                                   0.0); CeedChk(ierr);
63242ea3801Sjeremylt       }
6331d102b48SJeremy L Thompson       // Set Outputs
6341d102b48SJeremy L Thompson       for (CeedInt out=0; out<numoutputfields; out++) {
6351d102b48SJeremy L Thompson         // Get output vector
6361d102b48SJeremy L Thompson         ierr = CeedOperatorFieldGetVector(opoutputfields[out], &vec);
6371d102b48SJeremy L Thompson         CeedChk(ierr);
6381d102b48SJeremy L Thompson         // Check if active output
6391d102b48SJeremy L Thompson         if (vec == CEED_VECTOR_ACTIVE) {
6401d102b48SJeremy L Thompson           CeedVectorSetArray(impl->qvecsout[out], CEED_MEM_HOST,
6411d102b48SJeremy L Thompson                              CEED_USE_POINTER, a); CeedChk(ierr);
6421d102b48SJeremy L Thompson           ierr = CeedQFunctionFieldGetSize(qfoutputfields[out], &size);
6431d102b48SJeremy L Thompson           CeedChk(ierr);
6441d102b48SJeremy L Thompson           a += size*Q*blksize; // Advance the pointer by the size of the output
6451d102b48SJeremy L Thompson         }
6461d102b48SJeremy L Thompson       }
6471d102b48SJeremy L Thompson       // Apply QFunction
6481d102b48SJeremy L Thompson       ierr = CeedQFunctionApply(qf, Q*blksize, impl->qvecsin, impl->qvecsout);
649d1bcdac9Sjeremylt       CeedChk(ierr);
6504a2e7687Sjeremylt     }
6514a2e7687Sjeremylt   }
6524a2e7687Sjeremylt 
6531d102b48SJeremy L Thompson   // Un-set output Qvecs to prevent accidental overwrite of Assembled
6541d102b48SJeremy L Thompson   for (CeedInt out=0; out<numoutputfields; out++) {
6551d102b48SJeremy L Thompson     // Get output vector
6561d102b48SJeremy L Thompson     ierr = CeedOperatorFieldGetVector(opoutputfields[out], &vec);
6571d102b48SJeremy L Thompson     CeedChk(ierr);
6581d102b48SJeremy L Thompson     // Check if active output
6591d102b48SJeremy L Thompson     if (vec == CEED_VECTOR_ACTIVE) {
6601d102b48SJeremy L Thompson       CeedVectorSetArray(impl->qvecsout[out], CEED_MEM_HOST, CEED_COPY_VALUES,
6611d102b48SJeremy L Thompson                          NULL); CeedChk(ierr);
6621d102b48SJeremy L Thompson     }
6631d102b48SJeremy L Thompson   }
6641d102b48SJeremy L Thompson 
6651d102b48SJeremy L Thompson   // Restore input arrays
6661d102b48SJeremy L Thompson   ierr = CeedOperatorRestoreInputs_Blocked(numinputfields, qfinputfields,
6677f823360Sjeremylt          opinputfields, true, impl); CeedChk(ierr);
6681d102b48SJeremy L Thompson 
6691d102b48SJeremy L Thompson   // Output blocked restriction
6701d102b48SJeremy L Thompson   ierr = CeedVectorRestoreArray(lvec, &a); CeedChk(ierr);
6711d102b48SJeremy L Thompson   ierr = CeedVectorSetValue(*assembled, 0.0); CeedChk(ierr);
6721d102b48SJeremy L Thompson   CeedElemRestriction blkrstr;
6737509a596Sjeremylt   ierr = CeedElemRestrictionCreateBlockedStrided(ceed, numelements, Q, blksize,
674d979a051Sjeremylt          numactivein*numactiveout, numactivein*numactiveout*numelements*Q,
675d979a051Sjeremylt          strides, &blkrstr); CeedChk(ierr);
676a8d32208Sjeremylt   ierr = CeedElemRestrictionApply(blkrstr, CEED_TRANSPOSE, lvec, *assembled,
677a8d32208Sjeremylt                                   request); CeedChk(ierr);
6781d102b48SJeremy L Thompson 
6791d102b48SJeremy L Thompson   // Cleanup
68042ea3801Sjeremylt   for (CeedInt i=0; i<numactivein; i++) {
68142ea3801Sjeremylt     ierr = CeedVectorDestroy(&activein[i]); CeedChk(ierr);
68242ea3801Sjeremylt   }
6831d102b48SJeremy L Thompson   ierr = CeedFree(&activein); CeedChk(ierr);
6841d102b48SJeremy L Thompson   ierr = CeedVectorDestroy(&lvec); CeedChk(ierr);
6851d102b48SJeremy L Thompson   ierr = CeedElemRestrictionDestroy(&blkrstr); CeedChk(ierr);
6861d102b48SJeremy L Thompson 
6874a2e7687Sjeremylt   return 0;
6884a2e7687Sjeremylt }
6894a2e7687Sjeremylt 
690f10650afSjeremylt //------------------------------------------------------------------------------
691f10650afSjeremylt // Operator Destroy
692f10650afSjeremylt //------------------------------------------------------------------------------
693f10650afSjeremylt static int CeedOperatorDestroy_Blocked(CeedOperator op) {
694f10650afSjeremylt   int ierr;
695f10650afSjeremylt   CeedOperator_Blocked *impl;
696f10650afSjeremylt   ierr = CeedOperatorGetData(op, (void *)&impl); CeedChk(ierr);
697f10650afSjeremylt 
698f10650afSjeremylt   for (CeedInt i=0; i<impl->numein+impl->numeout; i++) {
699f10650afSjeremylt     ierr = CeedElemRestrictionDestroy(&impl->blkrestr[i]); CeedChk(ierr);
700f10650afSjeremylt     ierr = CeedVectorDestroy(&impl->evecs[i]); CeedChk(ierr);
701f10650afSjeremylt   }
702f10650afSjeremylt   ierr = CeedFree(&impl->blkrestr); CeedChk(ierr);
703f10650afSjeremylt   ierr = CeedFree(&impl->evecs); CeedChk(ierr);
704f10650afSjeremylt   ierr = CeedFree(&impl->edata); CeedChk(ierr);
705f10650afSjeremylt   ierr = CeedFree(&impl->inputstate); CeedChk(ierr);
706f10650afSjeremylt 
707f10650afSjeremylt   for (CeedInt i=0; i<impl->numein; i++) {
708f10650afSjeremylt     ierr = CeedVectorDestroy(&impl->evecsin[i]); CeedChk(ierr);
709f10650afSjeremylt     ierr = CeedVectorDestroy(&impl->qvecsin[i]); CeedChk(ierr);
710f10650afSjeremylt   }
711f10650afSjeremylt   ierr = CeedFree(&impl->evecsin); CeedChk(ierr);
712f10650afSjeremylt   ierr = CeedFree(&impl->qvecsin); CeedChk(ierr);
713f10650afSjeremylt 
714f10650afSjeremylt   for (CeedInt i=0; i<impl->numeout; i++) {
715f10650afSjeremylt     ierr = CeedVectorDestroy(&impl->evecsout[i]); CeedChk(ierr);
716f10650afSjeremylt     ierr = CeedVectorDestroy(&impl->qvecsout[i]); CeedChk(ierr);
717f10650afSjeremylt   }
718f10650afSjeremylt   ierr = CeedFree(&impl->evecsout); CeedChk(ierr);
719f10650afSjeremylt   ierr = CeedFree(&impl->qvecsout); CeedChk(ierr);
720f10650afSjeremylt 
721f10650afSjeremylt   ierr = CeedFree(&impl); CeedChk(ierr);
722f10650afSjeremylt   return 0;
723f10650afSjeremylt }
724f10650afSjeremylt 
725f10650afSjeremylt //------------------------------------------------------------------------------
726f10650afSjeremylt // Operator Create
727f10650afSjeremylt //------------------------------------------------------------------------------
7284a2e7687Sjeremylt int CeedOperatorCreate_Blocked(CeedOperator op) {
7294a2e7687Sjeremylt   int ierr;
730fe2413ffSjeremylt   Ceed ceed;
731fe2413ffSjeremylt   ierr = CeedOperatorGetCeed(op, &ceed); CeedChk(ierr);
7324ce2993fSjeremylt   CeedOperator_Blocked *impl;
7334a2e7687Sjeremylt 
7344a2e7687Sjeremylt   ierr = CeedCalloc(1, &impl); CeedChk(ierr);
735de686571SJeremy L Thompson   ierr = CeedOperatorSetData(op, (void *)&impl); CeedChk(ierr);
736fe2413ffSjeremylt 
7371d102b48SJeremy L Thompson   ierr = CeedSetBackendFunction(ceed, "Operator", op, "AssembleLinearQFunction",
7381d102b48SJeremy L Thompson                                 CeedOperatorAssembleLinearQFunction_Blocked);
7391d102b48SJeremy L Thompson   CeedChk(ierr);
740cae8b89aSjeremylt   ierr = CeedSetBackendFunction(ceed, "Operator", op, "ApplyAdd",
741fe2413ffSjeremylt                                 CeedOperatorApply_Blocked); CeedChk(ierr);
742fe2413ffSjeremylt   ierr = CeedSetBackendFunction(ceed, "Operator", op, "Destroy",
743fe2413ffSjeremylt                                 CeedOperatorDestroy_Blocked); CeedChk(ierr);
7444a2e7687Sjeremylt   return 0;
7454a2e7687Sjeremylt }
746f10650afSjeremylt //------------------------------------------------------------------------------
747