xref: /libCEED/rust/libceed-sys/c-src/backends/blocked/ceed-blocked-operator.c (revision e15f9bd09af0280c89b79924fa9af7dd2e3e30be)
14a2e7687Sjeremylt // Copyright (c) 2017-2018, Lawrence Livermore National Security, LLC.
24a2e7687Sjeremylt // Produced at the Lawrence Livermore National Laboratory. LLNL-CODE-734707.
34a2e7687Sjeremylt // All Rights reserved. See files LICENSE and NOTICE for details.
44a2e7687Sjeremylt //
54a2e7687Sjeremylt // This file is part of CEED, a collection of benchmarks, miniapps, software
64a2e7687Sjeremylt // libraries and APIs for efficient high-order finite element and spectral
74a2e7687Sjeremylt // element discretizations for exascale applications. For more information and
84a2e7687Sjeremylt // source code availability see http://github.com/ceed.
94a2e7687Sjeremylt //
104a2e7687Sjeremylt // The CEED research is supported by the Exascale Computing Project 17-SC-20-SC,
114a2e7687Sjeremylt // a collaborative effort of two U.S. Department of Energy organizations (Office
124a2e7687Sjeremylt // of Science and the National Nuclear Security Administration) responsible for
134a2e7687Sjeremylt // the planning and preparation of a capable exascale ecosystem, including
144a2e7687Sjeremylt // software, applications, hardware, advanced system engineering and early
154a2e7687Sjeremylt // testbed platforms, in support of the nation's exascale computing imperative.
164a2e7687Sjeremylt 
173d576824SJeremy L Thompson #include <ceed.h>
183d576824SJeremy L Thompson #include <ceed-backend.h>
193d576824SJeremy L Thompson #include <stdbool.h>
203d576824SJeremy L Thompson #include <stddef.h>
213d576824SJeremy L Thompson #include <stdint.h>
224a2e7687Sjeremylt #include "ceed-blocked.h"
234a2e7687Sjeremylt 
24f10650afSjeremylt //------------------------------------------------------------------------------
25f10650afSjeremylt // Setup Input/Output Fields
26f10650afSjeremylt //------------------------------------------------------------------------------
2789c6efa4Sjeremylt static int CeedOperatorSetupFields_Blocked(CeedQFunction qf,
2889c6efa4Sjeremylt     CeedOperator op, bool inOrOut,
294a2e7687Sjeremylt     CeedElemRestriction *blkrestr,
3091703d3fSjeremylt     CeedVector *fullevecs, CeedVector *evecs,
31aedaa0e5Sjeremylt     CeedVector *qvecs, CeedInt starte,
324a2e7687Sjeremylt     CeedInt numfields, CeedInt Q) {
334d537eeaSYohann   CeedInt dim, ierr, ncomp, size, P;
34aedaa0e5Sjeremylt   Ceed ceed;
35*e15f9bd0SJeremy L Thompson   ierr = CeedOperatorGetCeed(op, &ceed); CeedChkBackend(ierr);
36d1bcdac9Sjeremylt   CeedBasis basis;
37d1bcdac9Sjeremylt   CeedElemRestriction r;
38aedaa0e5Sjeremylt   CeedOperatorField *opfields;
39aedaa0e5Sjeremylt   CeedQFunctionField *qffields;
40fe2413ffSjeremylt   if (inOrOut) {
41aedaa0e5Sjeremylt     ierr = CeedOperatorGetFields(op, NULL, &opfields);
42*e15f9bd0SJeremy L Thompson     CeedChkBackend(ierr);
43aedaa0e5Sjeremylt     ierr = CeedQFunctionGetFields(qf, NULL, &qffields);
44*e15f9bd0SJeremy L Thompson     CeedChkBackend(ierr);
45fe2413ffSjeremylt   } else {
46aedaa0e5Sjeremylt     ierr = CeedOperatorGetFields(op, &opfields, NULL);
47*e15f9bd0SJeremy L Thompson     CeedChkBackend(ierr);
48aedaa0e5Sjeremylt     ierr = CeedQFunctionGetFields(qf, &qffields, NULL);
49*e15f9bd0SJeremy L Thompson     CeedChkBackend(ierr);
50fe2413ffSjeremylt   }
514a2e7687Sjeremylt   const CeedInt blksize = 8;
524a2e7687Sjeremylt 
534a2e7687Sjeremylt   // Loop over fields
544a2e7687Sjeremylt   for (CeedInt i=0; i<numfields; i++) {
55d1bcdac9Sjeremylt     CeedEvalMode emode;
56*e15f9bd0SJeremy L Thompson     ierr = CeedQFunctionFieldGetEvalMode(qffields[i], &emode); CeedChkBackend(ierr);
574a2e7687Sjeremylt 
584a2e7687Sjeremylt     if (emode != CEED_EVAL_WEIGHT) {
59aedaa0e5Sjeremylt       ierr = CeedOperatorFieldGetElemRestriction(opfields[i], &r);
60*e15f9bd0SJeremy L Thompson       CeedChkBackend(ierr);
61*e15f9bd0SJeremy L Thompson       ierr = CeedElemRestrictionGetCeed(r, &ceed); CeedChkBackend(ierr);
62d979a051Sjeremylt       CeedInt nelem, elemsize, lsize, compstride;
63*e15f9bd0SJeremy L Thompson       ierr = CeedElemRestrictionGetNumElements(r, &nelem); CeedChkBackend(ierr);
64*e15f9bd0SJeremy L Thompson       ierr = CeedElemRestrictionGetElementSize(r, &elemsize); CeedChkBackend(ierr);
65*e15f9bd0SJeremy L Thompson       ierr = CeedElemRestrictionGetLVectorSize(r, &lsize); CeedChkBackend(ierr);
66*e15f9bd0SJeremy L Thompson       ierr = CeedElemRestrictionGetNumComponents(r, &ncomp); CeedChkBackend(ierr);
67bd33150aSjeremylt 
683ac43b2cSJeremy L Thompson       bool strided;
69*e15f9bd0SJeremy L Thompson       ierr = CeedElemRestrictionIsStrided(r, &strided); CeedChkBackend(ierr);
703ac43b2cSJeremy L Thompson       if (strided) {
713ac43b2cSJeremy L Thompson         CeedInt strides[3];
72*e15f9bd0SJeremy L Thompson         ierr = CeedElemRestrictionGetStrides(r, &strides); CeedChkBackend(ierr);
733ac43b2cSJeremy L Thompson         ierr = CeedElemRestrictionCreateBlockedStrided(ceed, nelem, elemsize,
743ac43b2cSJeremy L Thompson                blksize, ncomp, lsize, strides, &blkrestr[i+starte]);
75*e15f9bd0SJeremy L Thompson         CeedChkBackend(ierr);
763ac43b2cSJeremy L Thompson       } else {
77bd33150aSjeremylt         const CeedInt *offsets = NULL;
78bd33150aSjeremylt         ierr = CeedElemRestrictionGetOffsets(r, CEED_MEM_HOST, &offsets);
79*e15f9bd0SJeremy L Thompson         CeedChkBackend(ierr);
80*e15f9bd0SJeremy L Thompson         ierr = CeedElemRestrictionGetCompStride(r, &compstride); CeedChkBackend(ierr);
81d979a051Sjeremylt         ierr = CeedElemRestrictionCreateBlocked(ceed, nelem, elemsize,
82d979a051Sjeremylt                                                 blksize, ncomp, compstride,
83d979a051Sjeremylt                                                 lsize, CEED_MEM_HOST,
84bd33150aSjeremylt                                                 CEED_COPY_VALUES, offsets,
857509a596Sjeremylt                                                 &blkrestr[i+starte]);
86*e15f9bd0SJeremy L Thompson         CeedChkBackend(ierr);
87*e15f9bd0SJeremy L Thompson         ierr = CeedElemRestrictionRestoreOffsets(r, &offsets); CeedChkBackend(ierr);
883ac43b2cSJeremy L Thompson       }
89aedaa0e5Sjeremylt       ierr = CeedElemRestrictionCreateVector(blkrestr[i+starte], NULL,
9091703d3fSjeremylt                                              &fullevecs[i+starte]);
91*e15f9bd0SJeremy L Thompson       CeedChkBackend(ierr);
924a2e7687Sjeremylt     }
934a2e7687Sjeremylt 
944a2e7687Sjeremylt     switch(emode) {
954a2e7687Sjeremylt     case CEED_EVAL_NONE:
96*e15f9bd0SJeremy L Thompson       ierr = CeedQFunctionFieldGetSize(qffields[i], &size); CeedChkBackend(ierr);
97*e15f9bd0SJeremy L Thompson       ierr = CeedVectorCreate(ceed, Q*size*blksize, &qvecs[i]); CeedChkBackend(ierr);
98aedaa0e5Sjeremylt       break;
99aedaa0e5Sjeremylt     case CEED_EVAL_INTERP:
100*e15f9bd0SJeremy L Thompson       ierr = CeedQFunctionFieldGetSize(qffields[i], &size); CeedChkBackend(ierr);
1014d1cd9fcSJeremy L Thompson       ierr = CeedElemRestrictionGetElementSize(r, &P);
102*e15f9bd0SJeremy L Thompson       CeedChkBackend(ierr);
103*e15f9bd0SJeremy L Thompson       ierr = CeedVectorCreate(ceed, P*size*blksize, &evecs[i]); CeedChkBackend(ierr);
104*e15f9bd0SJeremy L Thompson       ierr = CeedVectorCreate(ceed, Q*size*blksize, &qvecs[i]); CeedChkBackend(ierr);
1054a2e7687Sjeremylt       break;
1064a2e7687Sjeremylt     case CEED_EVAL_GRAD:
107*e15f9bd0SJeremy L Thompson       ierr = CeedOperatorFieldGetBasis(opfields[i], &basis); CeedChkBackend(ierr);
108*e15f9bd0SJeremy L Thompson       ierr = CeedQFunctionFieldGetSize(qffields[i], &size); CeedChkBackend(ierr);
109*e15f9bd0SJeremy L Thompson       ierr = CeedBasisGetDimension(basis, &dim); CeedChkBackend(ierr);
1104d1cd9fcSJeremy L Thompson       ierr = CeedElemRestrictionGetElementSize(r, &P);
111*e15f9bd0SJeremy L Thompson       CeedChkBackend(ierr);
112*e15f9bd0SJeremy L Thompson       ierr = CeedVectorCreate(ceed, P*size/dim*blksize, &evecs[i]);
113*e15f9bd0SJeremy L Thompson       CeedChkBackend(ierr);
114*e15f9bd0SJeremy L Thompson       ierr = CeedVectorCreate(ceed, Q*size*blksize, &qvecs[i]); CeedChkBackend(ierr);
1154a2e7687Sjeremylt       break;
1164a2e7687Sjeremylt     case CEED_EVAL_WEIGHT: // Only on input fields
117*e15f9bd0SJeremy L Thompson       ierr = CeedOperatorFieldGetBasis(opfields[i], &basis); CeedChkBackend(ierr);
118*e15f9bd0SJeremy L Thompson       ierr = CeedVectorCreate(ceed, Q*blksize, &qvecs[i]); CeedChkBackend(ierr);
119d1bcdac9Sjeremylt       ierr = CeedBasisApply(basis, blksize, CEED_NOTRANSPOSE,
120a7b7f929Sjeremylt                             CEED_EVAL_WEIGHT, CEED_VECTOR_NONE, qvecs[i]);
121*e15f9bd0SJeremy L Thompson       CeedChkBackend(ierr);
122aedaa0e5Sjeremylt 
1234a2e7687Sjeremylt       break;
1244a2e7687Sjeremylt     case CEED_EVAL_DIV:
1254d537eeaSYohann       break; // Not implemented
1264a2e7687Sjeremylt     case CEED_EVAL_CURL:
1274d537eeaSYohann       break; // Not implemented
1284a2e7687Sjeremylt     }
1294a2e7687Sjeremylt   }
130*e15f9bd0SJeremy L Thompson   return CEED_ERROR_SUCCESS;
1314a2e7687Sjeremylt }
1324a2e7687Sjeremylt 
133f10650afSjeremylt //------------------------------------------------------------------------------
134f10650afSjeremylt // Setup Operator
135f10650afSjeremylt //------------------------------------------------------------------------------
1364a2e7687Sjeremylt static int CeedOperatorSetup_Blocked(CeedOperator op) {
1374a2e7687Sjeremylt   int ierr;
1384ce2993fSjeremylt   bool setupdone;
139*e15f9bd0SJeremy L Thompson   ierr = CeedOperatorIsSetupDone(op, &setupdone); CeedChkBackend(ierr);
140*e15f9bd0SJeremy L Thompson   if (setupdone) return CEED_ERROR_SUCCESS;
141aedaa0e5Sjeremylt   Ceed ceed;
142*e15f9bd0SJeremy L Thompson   ierr = CeedOperatorGetCeed(op, &ceed); CeedChkBackend(ierr);
1434ce2993fSjeremylt   CeedOperator_Blocked *impl;
144*e15f9bd0SJeremy L Thompson   ierr = CeedOperatorGetData(op, &impl); CeedChkBackend(ierr);
1454ce2993fSjeremylt   CeedQFunction qf;
146*e15f9bd0SJeremy L Thompson   ierr = CeedOperatorGetQFunction(op, &qf); CeedChkBackend(ierr);
1474ce2993fSjeremylt   CeedInt Q, numinputfields, numoutputfields;
148*e15f9bd0SJeremy L Thompson   ierr = CeedOperatorGetNumQuadraturePoints(op, &Q); CeedChkBackend(ierr);
149*e15f9bd0SJeremy L Thompson   ierr = CeedQFunctionIsIdentity(qf, &impl->identityqf); CeedChkBackend(ierr);
1504a2e7687Sjeremylt   ierr= CeedQFunctionGetNumArgs(qf, &numinputfields, &numoutputfields);
151*e15f9bd0SJeremy L Thompson   CeedChkBackend(ierr);
152d1bcdac9Sjeremylt   CeedOperatorField *opinputfields, *opoutputfields;
153d1bcdac9Sjeremylt   ierr = CeedOperatorGetFields(op, &opinputfields, &opoutputfields);
154*e15f9bd0SJeremy L Thompson   CeedChkBackend(ierr);
155d1bcdac9Sjeremylt   CeedQFunctionField *qfinputfields, *qfoutputfields;
156d1bcdac9Sjeremylt   ierr = CeedQFunctionGetFields(qf, &qfinputfields, &qfoutputfields);
157*e15f9bd0SJeremy L Thompson   CeedChkBackend(ierr);
1584a2e7687Sjeremylt 
1594a2e7687Sjeremylt   // Allocate
160aedaa0e5Sjeremylt   ierr = CeedCalloc(numinputfields + numoutputfields, &impl->blkrestr);
161*e15f9bd0SJeremy L Thompson   CeedChkBackend(ierr);
162aedaa0e5Sjeremylt   ierr = CeedCalloc(numinputfields + numoutputfields, &impl->evecs);
163*e15f9bd0SJeremy L Thompson   CeedChkBackend(ierr);
164aedaa0e5Sjeremylt   ierr = CeedCalloc(numinputfields + numoutputfields, &impl->edata);
165*e15f9bd0SJeremy L Thompson   CeedChkBackend(ierr);
1664a2e7687Sjeremylt 
167*e15f9bd0SJeremy L Thompson   ierr = CeedCalloc(16, &impl->inputstate); CeedChkBackend(ierr);
168*e15f9bd0SJeremy L Thompson   ierr = CeedCalloc(16, &impl->evecsin); CeedChkBackend(ierr);
169*e15f9bd0SJeremy L Thompson   ierr = CeedCalloc(16, &impl->evecsout); CeedChkBackend(ierr);
170*e15f9bd0SJeremy L Thompson   ierr = CeedCalloc(16, &impl->qvecsin); CeedChkBackend(ierr);
171*e15f9bd0SJeremy L Thompson   ierr = CeedCalloc(16, &impl->qvecsout); CeedChkBackend(ierr);
1724a2e7687Sjeremylt 
173aedaa0e5Sjeremylt   impl->numein = numinputfields; impl->numeout = numoutputfields;
174aedaa0e5Sjeremylt 
1754a2e7687Sjeremylt   // Set up infield and outfield pointer arrays
1764a2e7687Sjeremylt   // Infields
177aedaa0e5Sjeremylt   ierr = CeedOperatorSetupFields_Blocked(qf, op, 0, impl->blkrestr,
17891703d3fSjeremylt                                          impl->evecs, impl->evecsin,
17991703d3fSjeremylt                                          impl->qvecsin, 0,
180aedaa0e5Sjeremylt                                          numinputfields, Q);
181*e15f9bd0SJeremy L Thompson   CeedChkBackend(ierr);
1824a2e7687Sjeremylt   // Outfields
183aedaa0e5Sjeremylt   ierr = CeedOperatorSetupFields_Blocked(qf, op, 1, impl->blkrestr,
18491703d3fSjeremylt                                          impl->evecs, impl->evecsout,
18591703d3fSjeremylt                                          impl->qvecsout, numinputfields,
18691703d3fSjeremylt                                          numoutputfields, Q);
187*e15f9bd0SJeremy L Thompson   CeedChkBackend(ierr);
188aedaa0e5Sjeremylt 
18916911fdaSjeremylt   // Identity QFunctions
19016911fdaSjeremylt   if (impl->identityqf) {
19116911fdaSjeremylt     CeedEvalMode inmode, outmode;
19216911fdaSjeremylt     CeedQFunctionField *infields, *outfields;
193*e15f9bd0SJeremy L Thompson     ierr = CeedQFunctionGetFields(qf, &infields, &outfields); CeedChkBackend(ierr);
19416911fdaSjeremylt 
19516911fdaSjeremylt     for (CeedInt i=0; i<numinputfields; i++) {
19616911fdaSjeremylt       ierr = CeedQFunctionFieldGetEvalMode(infields[i], &inmode);
197*e15f9bd0SJeremy L Thompson       CeedChkBackend(ierr);
19816911fdaSjeremylt       ierr = CeedQFunctionFieldGetEvalMode(outfields[i], &outmode);
199*e15f9bd0SJeremy L Thompson       CeedChkBackend(ierr);
20016911fdaSjeremylt 
201*e15f9bd0SJeremy L Thompson       ierr = CeedVectorDestroy(&impl->qvecsout[i]); CeedChkBackend(ierr);
20216911fdaSjeremylt       impl->qvecsout[i] = impl->qvecsin[i];
203*e15f9bd0SJeremy L Thompson       ierr = CeedVectorAddReference(impl->qvecsin[i]); CeedChkBackend(ierr);
20416911fdaSjeremylt     }
20516911fdaSjeremylt   }
20616911fdaSjeremylt 
207*e15f9bd0SJeremy L Thompson   ierr = CeedOperatorSetSetupDone(op); CeedChkBackend(ierr);
2084a2e7687Sjeremylt 
209*e15f9bd0SJeremy L Thompson   return CEED_ERROR_SUCCESS;
2104a2e7687Sjeremylt }
2114a2e7687Sjeremylt 
212f10650afSjeremylt //------------------------------------------------------------------------------
213f10650afSjeremylt // Setup Operator Inputs
214f10650afSjeremylt //------------------------------------------------------------------------------
2151d102b48SJeremy L Thompson static inline int CeedOperatorSetupInputs_Blocked(CeedInt numinputfields,
2161d102b48SJeremy L Thompson     CeedQFunctionField *qfinputfields, CeedOperatorField *opinputfields,
2171d102b48SJeremy L Thompson     CeedVector invec, bool skipactive, CeedOperator_Blocked *impl,
21889c6efa4Sjeremylt     CeedRequest *request) {
2191d102b48SJeremy L Thompson   CeedInt ierr;
220d1bcdac9Sjeremylt   CeedEvalMode emode;
221d1bcdac9Sjeremylt   CeedVector vec;
22216c359e6Sjeremylt   uint64_t state;
2234a2e7687Sjeremylt 
2244a2e7687Sjeremylt   for (CeedInt i=0; i<numinputfields; i++) {
2251d102b48SJeremy L Thompson     // Get input vector
226*e15f9bd0SJeremy L Thompson     ierr = CeedOperatorFieldGetVector(opinputfields[i], &vec); CeedChkBackend(ierr);
2271d102b48SJeremy L Thompson     if (vec == CEED_VECTOR_ACTIVE) {
2281d102b48SJeremy L Thompson       if (skipactive)
2291d102b48SJeremy L Thompson         continue;
2301d102b48SJeremy L Thompson       else
2311d102b48SJeremy L Thompson         vec = invec;
2321d102b48SJeremy L Thompson     }
2331d102b48SJeremy L Thompson 
234d1bcdac9Sjeremylt     ierr = CeedQFunctionFieldGetEvalMode(qfinputfields[i], &emode);
235*e15f9bd0SJeremy L Thompson     CeedChkBackend(ierr);
2364a2e7687Sjeremylt     if (emode == CEED_EVAL_WEIGHT) { // Skip
2374a2e7687Sjeremylt     } else {
2384a2e7687Sjeremylt       // Restrict
239*e15f9bd0SJeremy L Thompson       ierr = CeedVectorGetState(vec, &state); CeedChkBackend(ierr);
24089c6efa4Sjeremylt       if (state != impl->inputstate[i] || vec == invec) {
2414a2e7687Sjeremylt         ierr = CeedElemRestrictionApply(impl->blkrestr[i], CEED_NOTRANSPOSE,
242a8d32208Sjeremylt                                         vec, impl->evecs[i], request);
243*e15f9bd0SJeremy L Thompson         CeedChkBackend(ierr);
24416c359e6Sjeremylt         impl->inputstate[i] = state;
24516c359e6Sjeremylt       }
2464a2e7687Sjeremylt       // Get evec
2474a2e7687Sjeremylt       ierr = CeedVectorGetArrayRead(impl->evecs[i], CEED_MEM_HOST,
2484a2e7687Sjeremylt                                     (const CeedScalar **) &impl->edata[i]);
249*e15f9bd0SJeremy L Thompson       CeedChkBackend(ierr);
2504a2e7687Sjeremylt     }
2514a2e7687Sjeremylt   }
252*e15f9bd0SJeremy L Thompson   return CEED_ERROR_SUCCESS;
2534a2e7687Sjeremylt }
2544a2e7687Sjeremylt 
255f10650afSjeremylt //------------------------------------------------------------------------------
256f10650afSjeremylt // Input Basis Action
257f10650afSjeremylt //------------------------------------------------------------------------------
2581d102b48SJeremy L Thompson static inline int CeedOperatorInputBasis_Blocked(CeedInt e, CeedInt Q,
2591d102b48SJeremy L Thompson     CeedQFunctionField *qfinputfields, CeedOperatorField *opinputfields,
2601d102b48SJeremy L Thompson     CeedInt numinputfields, CeedInt blksize, bool skipactive,
2611d102b48SJeremy L Thompson     CeedOperator_Blocked *impl) {
2621d102b48SJeremy L Thompson   CeedInt ierr;
2631d102b48SJeremy L Thompson   CeedInt dim, elemsize, size;
2641d102b48SJeremy L Thompson   CeedElemRestriction Erestrict;
2651d102b48SJeremy L Thompson   CeedEvalMode emode;
2661d102b48SJeremy L Thompson   CeedBasis basis;
2671d102b48SJeremy L Thompson 
2684a2e7687Sjeremylt   for (CeedInt i=0; i<numinputfields; i++) {
2691d102b48SJeremy L Thompson     // Skip active input
2701d102b48SJeremy L Thompson     if (skipactive) {
2711d102b48SJeremy L Thompson       CeedVector vec;
272*e15f9bd0SJeremy L Thompson       ierr = CeedOperatorFieldGetVector(opinputfields[i], &vec); CeedChkBackend(ierr);
2731d102b48SJeremy L Thompson       if (vec == CEED_VECTOR_ACTIVE)
2741d102b48SJeremy L Thompson         continue;
2751d102b48SJeremy L Thompson     }
2761d102b48SJeremy L Thompson 
2774d537eeaSYohann     // Get elemsize, emode, size
278d1bcdac9Sjeremylt     ierr = CeedOperatorFieldGetElemRestriction(opinputfields[i], &Erestrict);
279*e15f9bd0SJeremy L Thompson     CeedChkBackend(ierr);
280d1bcdac9Sjeremylt     ierr = CeedElemRestrictionGetElementSize(Erestrict, &elemsize);
281*e15f9bd0SJeremy L Thompson     CeedChkBackend(ierr);
282d1bcdac9Sjeremylt     ierr = CeedQFunctionFieldGetEvalMode(qfinputfields[i], &emode);
283*e15f9bd0SJeremy L Thompson     CeedChkBackend(ierr);
284*e15f9bd0SJeremy L Thompson     ierr = CeedQFunctionFieldGetSize(qfinputfields[i], &size); CeedChkBackend(ierr);
2854a2e7687Sjeremylt     // Basis action
2864a2e7687Sjeremylt     switch(emode) {
2874a2e7687Sjeremylt     case CEED_EVAL_NONE:
288aedaa0e5Sjeremylt       ierr = CeedVectorSetArray(impl->qvecsin[i], CEED_MEM_HOST,
289aedaa0e5Sjeremylt                                 CEED_USE_POINTER,
290*e15f9bd0SJeremy L Thompson                                 &impl->edata[i][e*Q*size]); CeedChkBackend(ierr);
2914a2e7687Sjeremylt       break;
2924a2e7687Sjeremylt     case CEED_EVAL_INTERP:
293*e15f9bd0SJeremy L Thompson       ierr = CeedOperatorFieldGetBasis(opinputfields[i], &basis);
294*e15f9bd0SJeremy L Thompson       CeedChkBackend(ierr);
29591703d3fSjeremylt       ierr = CeedVectorSetArray(impl->evecsin[i], CEED_MEM_HOST,
296aedaa0e5Sjeremylt                                 CEED_USE_POINTER,
2974d537eeaSYohann                                 &impl->edata[i][e*elemsize*size]);
298*e15f9bd0SJeremy L Thompson       CeedChkBackend(ierr);
299d1bcdac9Sjeremylt       ierr = CeedBasisApply(basis, blksize, CEED_NOTRANSPOSE,
30091703d3fSjeremylt                             CEED_EVAL_INTERP, impl->evecsin[i],
301*e15f9bd0SJeremy L Thompson                             impl->qvecsin[i]); CeedChkBackend(ierr);
3024a2e7687Sjeremylt       break;
3034a2e7687Sjeremylt     case CEED_EVAL_GRAD:
304*e15f9bd0SJeremy L Thompson       ierr = CeedOperatorFieldGetBasis(opinputfields[i], &basis);
305*e15f9bd0SJeremy L Thompson       CeedChkBackend(ierr);
306*e15f9bd0SJeremy L Thompson       ierr = CeedBasisGetDimension(basis, &dim); CeedChkBackend(ierr);
30791703d3fSjeremylt       ierr = CeedVectorSetArray(impl->evecsin[i], CEED_MEM_HOST,
308aedaa0e5Sjeremylt                                 CEED_USE_POINTER,
3094d537eeaSYohann                                 &impl->edata[i][e*elemsize*size/dim]);
310*e15f9bd0SJeremy L Thompson       CeedChkBackend(ierr);
311d1bcdac9Sjeremylt       ierr = CeedBasisApply(basis, blksize, CEED_NOTRANSPOSE,
31291703d3fSjeremylt                             CEED_EVAL_GRAD, impl->evecsin[i],
313*e15f9bd0SJeremy L Thompson                             impl->qvecsin[i]); CeedChkBackend(ierr);
3144a2e7687Sjeremylt       break;
3154a2e7687Sjeremylt     case CEED_EVAL_WEIGHT:
3164a2e7687Sjeremylt       break;  // No action
317bbfacfcdSjeremylt     // LCOV_EXCL_START
3184a2e7687Sjeremylt     case CEED_EVAL_DIV:
3191d102b48SJeremy L Thompson     case CEED_EVAL_CURL: {
3201d102b48SJeremy L Thompson       ierr = CeedOperatorFieldGetBasis(opinputfields[i], &basis);
321*e15f9bd0SJeremy L Thompson       CeedChkBackend(ierr);
3221d102b48SJeremy L Thompson       Ceed ceed;
323*e15f9bd0SJeremy L Thompson       ierr = CeedBasisGetCeed(basis, &ceed); CeedChkBackend(ierr);
324*e15f9bd0SJeremy L Thompson       return CeedError(ceed, CEED_ERROR_BACKEND,
325*e15f9bd0SJeremy L Thompson                        "Ceed evaluation mode not implemented");
326bbfacfcdSjeremylt       // LCOV_EXCL_STOP
3274a2e7687Sjeremylt     }
3284a2e7687Sjeremylt     }
32989c6efa4Sjeremylt   }
330*e15f9bd0SJeremy L Thompson   return CEED_ERROR_SUCCESS;
33189c6efa4Sjeremylt }
3324a2e7687Sjeremylt 
333f10650afSjeremylt //------------------------------------------------------------------------------
334f10650afSjeremylt // Output Basis Action
335f10650afSjeremylt //------------------------------------------------------------------------------
3361d102b48SJeremy L Thompson static inline int CeedOperatorOutputBasis_Blocked(CeedInt e, CeedInt Q,
3371d102b48SJeremy L Thompson     CeedQFunctionField *qfoutputfields, CeedOperatorField *opoutputfields,
3381d102b48SJeremy L Thompson     CeedInt blksize, CeedInt numinputfields, CeedInt numoutputfields,
3391d102b48SJeremy L Thompson     CeedOperator op, CeedOperator_Blocked *impl) {
3401d102b48SJeremy L Thompson   CeedInt ierr;
3411d102b48SJeremy L Thompson   CeedInt dim, elemsize, size;
3421d102b48SJeremy L Thompson   CeedElemRestriction Erestrict;
3431d102b48SJeremy L Thompson   CeedEvalMode emode;
3441d102b48SJeremy L Thompson   CeedBasis basis;
3451d102b48SJeremy L Thompson 
3464a2e7687Sjeremylt   for (CeedInt i=0; i<numoutputfields; i++) {
3474d537eeaSYohann     // Get elemsize, emode, size
348d1bcdac9Sjeremylt     ierr = CeedOperatorFieldGetElemRestriction(opoutputfields[i], &Erestrict);
349*e15f9bd0SJeremy L Thompson     CeedChkBackend(ierr);
35089c6efa4Sjeremylt     ierr = CeedElemRestrictionGetElementSize(Erestrict, &elemsize);
351*e15f9bd0SJeremy L Thompson     CeedChkBackend(ierr);
352d1bcdac9Sjeremylt     ierr = CeedQFunctionFieldGetEvalMode(qfoutputfields[i], &emode);
353*e15f9bd0SJeremy L Thompson     CeedChkBackend(ierr);
354*e15f9bd0SJeremy L Thompson     ierr = CeedQFunctionFieldGetSize(qfoutputfields[i], &size);
355*e15f9bd0SJeremy L Thompson     CeedChkBackend(ierr);
3564a2e7687Sjeremylt     // Basis action
3574a2e7687Sjeremylt     switch(emode) {
3584a2e7687Sjeremylt     case CEED_EVAL_NONE:
3594a2e7687Sjeremylt       break; // No action
3604a2e7687Sjeremylt     case CEED_EVAL_INTERP:
361d1bcdac9Sjeremylt       ierr = CeedOperatorFieldGetBasis(opoutputfields[i], &basis);
362*e15f9bd0SJeremy L Thompson       CeedChkBackend(ierr);
36389c6efa4Sjeremylt       ierr = CeedVectorSetArray(impl->evecsout[i], CEED_MEM_HOST,
36489c6efa4Sjeremylt                                 CEED_USE_POINTER,
3654d537eeaSYohann                                 &impl->edata[i + numinputfields][e*elemsize*size]);
366*e15f9bd0SJeremy L Thompson       CeedChkBackend(ierr);
367aedaa0e5Sjeremylt       ierr = CeedBasisApply(basis, blksize, CEED_TRANSPOSE,
368aedaa0e5Sjeremylt                             CEED_EVAL_INTERP, impl->qvecsout[i],
369*e15f9bd0SJeremy L Thompson                             impl->evecsout[i]); CeedChkBackend(ierr);
3704a2e7687Sjeremylt       break;
3714a2e7687Sjeremylt     case CEED_EVAL_GRAD:
372d1bcdac9Sjeremylt       ierr = CeedOperatorFieldGetBasis(opoutputfields[i], &basis);
373*e15f9bd0SJeremy L Thompson       CeedChkBackend(ierr);
374*e15f9bd0SJeremy L Thompson       ierr = CeedBasisGetDimension(basis, &dim); CeedChkBackend(ierr);
37589c6efa4Sjeremylt       ierr = CeedVectorSetArray(impl->evecsout[i], CEED_MEM_HOST,
37689c6efa4Sjeremylt                                 CEED_USE_POINTER,
3774d537eeaSYohann                                 &impl->edata[i + numinputfields][e*elemsize*size/dim]);
378*e15f9bd0SJeremy L Thompson       CeedChkBackend(ierr);
379d1bcdac9Sjeremylt       ierr = CeedBasisApply(basis, blksize, CEED_TRANSPOSE,
380aedaa0e5Sjeremylt                             CEED_EVAL_GRAD, impl->qvecsout[i],
381*e15f9bd0SJeremy L Thompson                             impl->evecsout[i]); CeedChkBackend(ierr);
3824a2e7687Sjeremylt       break;
383c042f62fSJeremy L Thompson     // LCOV_EXCL_START
384bbfacfcdSjeremylt     case CEED_EVAL_WEIGHT: {
3854ce2993fSjeremylt       Ceed ceed;
386*e15f9bd0SJeremy L Thompson       ierr = CeedOperatorGetCeed(op, &ceed); CeedChkBackend(ierr);
387*e15f9bd0SJeremy L Thompson       return CeedError(ceed, CEED_ERROR_BACKEND,
388*e15f9bd0SJeremy L Thompson                        "CEED_EVAL_WEIGHT cannot be an output "
3891d102b48SJeremy L Thompson                        "evaluation mode");
3904ce2993fSjeremylt     }
3914a2e7687Sjeremylt     case CEED_EVAL_DIV:
3921d102b48SJeremy L Thompson     case CEED_EVAL_CURL: {
3931d102b48SJeremy L Thompson       Ceed ceed;
394*e15f9bd0SJeremy L Thompson       ierr = CeedOperatorGetCeed(op, &ceed); CeedChkBackend(ierr);
395*e15f9bd0SJeremy L Thompson       return CeedError(ceed, CEED_ERROR_BACKEND,
396*e15f9bd0SJeremy L Thompson                        "Ceed evaluation mode not implemented");
397bbfacfcdSjeremylt       // LCOV_EXCL_STOP
3984a2e7687Sjeremylt     }
39989c6efa4Sjeremylt     }
40089c6efa4Sjeremylt   }
401*e15f9bd0SJeremy L Thompson   return CEED_ERROR_SUCCESS;
4021d102b48SJeremy L Thompson }
4031d102b48SJeremy L Thompson 
404f10650afSjeremylt //------------------------------------------------------------------------------
405f10650afSjeremylt // Restore Input Vectors
406f10650afSjeremylt //------------------------------------------------------------------------------
4071d102b48SJeremy L Thompson static inline int CeedOperatorRestoreInputs_Blocked(CeedInt numinputfields,
4081d102b48SJeremy L Thompson     CeedQFunctionField *qfinputfields, CeedOperatorField *opinputfields,
4091d102b48SJeremy L Thompson     bool skipactive, CeedOperator_Blocked *impl) {
4101d102b48SJeremy L Thompson   CeedInt ierr;
4111d102b48SJeremy L Thompson   CeedEvalMode emode;
4121d102b48SJeremy L Thompson 
4131d102b48SJeremy L Thompson   for (CeedInt i=0; i<numinputfields; i++) {
4141d102b48SJeremy L Thompson     // Skip active inputs
4151d102b48SJeremy L Thompson     if (skipactive) {
4161d102b48SJeremy L Thompson       CeedVector vec;
417*e15f9bd0SJeremy L Thompson       ierr = CeedOperatorFieldGetVector(opinputfields[i], &vec); CeedChkBackend(ierr);
4181d102b48SJeremy L Thompson       if (vec == CEED_VECTOR_ACTIVE)
4191d102b48SJeremy L Thompson         continue;
4201d102b48SJeremy L Thompson     }
4211d102b48SJeremy L Thompson     ierr = CeedQFunctionFieldGetEvalMode(qfinputfields[i], &emode);
422*e15f9bd0SJeremy L Thompson     CeedChkBackend(ierr);
4231d102b48SJeremy L Thompson     if (emode == CEED_EVAL_WEIGHT) { // Skip
4241d102b48SJeremy L Thompson     } else {
4251d102b48SJeremy L Thompson       ierr = CeedVectorRestoreArrayRead(impl->evecs[i],
4261d102b48SJeremy L Thompson                                         (const CeedScalar **) &impl->edata[i]);
427*e15f9bd0SJeremy L Thompson       CeedChkBackend(ierr);
4281d102b48SJeremy L Thompson     }
4291d102b48SJeremy L Thompson   }
430*e15f9bd0SJeremy L Thompson   return CEED_ERROR_SUCCESS;
4311d102b48SJeremy L Thompson }
4321d102b48SJeremy L Thompson 
433f10650afSjeremylt //------------------------------------------------------------------------------
434f10650afSjeremylt // Operator Apply
435f10650afSjeremylt //------------------------------------------------------------------------------
43669af5e5fSJeremy L Thompson static int CeedOperatorApplyAdd_Blocked(CeedOperator op, CeedVector invec,
4371d102b48SJeremy L Thompson                                         CeedVector outvec,
4381d102b48SJeremy L Thompson                                         CeedRequest *request) {
4391d102b48SJeremy L Thompson   int ierr;
4401d102b48SJeremy L Thompson   CeedOperator_Blocked *impl;
441*e15f9bd0SJeremy L Thompson   ierr = CeedOperatorGetData(op, &impl); CeedChkBackend(ierr);
4421d102b48SJeremy L Thompson   const CeedInt blksize = 8;
4431d102b48SJeremy L Thompson   CeedInt Q, numinputfields, numoutputfields, numelements, size;
444*e15f9bd0SJeremy L Thompson   ierr = CeedOperatorGetNumElements(op, &numelements); CeedChkBackend(ierr);
445*e15f9bd0SJeremy L Thompson   ierr = CeedOperatorGetNumQuadraturePoints(op, &Q); CeedChkBackend(ierr);
4461d102b48SJeremy L Thompson   CeedInt nblks = (numelements/blksize) + !!(numelements%blksize);
4471d102b48SJeremy L Thompson   CeedQFunction qf;
448*e15f9bd0SJeremy L Thompson   ierr = CeedOperatorGetQFunction(op, &qf); CeedChkBackend(ierr);
4491d102b48SJeremy L Thompson   ierr= CeedQFunctionGetNumArgs(qf, &numinputfields, &numoutputfields);
450*e15f9bd0SJeremy L Thompson   CeedChkBackend(ierr);
4511d102b48SJeremy L Thompson   CeedOperatorField *opinputfields, *opoutputfields;
4521d102b48SJeremy L Thompson   ierr = CeedOperatorGetFields(op, &opinputfields, &opoutputfields);
453*e15f9bd0SJeremy L Thompson   CeedChkBackend(ierr);
4541d102b48SJeremy L Thompson   CeedQFunctionField *qfinputfields, *qfoutputfields;
4551d102b48SJeremy L Thompson   ierr = CeedQFunctionGetFields(qf, &qfinputfields, &qfoutputfields);
456*e15f9bd0SJeremy L Thompson   CeedChkBackend(ierr);
4571d102b48SJeremy L Thompson   CeedEvalMode emode;
4581d102b48SJeremy L Thompson   CeedVector vec;
4591d102b48SJeremy L Thompson 
4601d102b48SJeremy L Thompson   // Setup
461*e15f9bd0SJeremy L Thompson   ierr = CeedOperatorSetup_Blocked(op); CeedChkBackend(ierr);
4621d102b48SJeremy L Thompson 
4631d102b48SJeremy L Thompson   // Input Evecs and Restriction
4641d102b48SJeremy L Thompson   ierr = CeedOperatorSetupInputs_Blocked(numinputfields, qfinputfields,
4657f823360Sjeremylt                                          opinputfields, invec, false, impl,
466*e15f9bd0SJeremy L Thompson                                          request); CeedChkBackend(ierr);
4671d102b48SJeremy L Thompson 
4681d102b48SJeremy L Thompson   // Output Evecs
4691d102b48SJeremy L Thompson   for (CeedInt i=0; i<numoutputfields; i++) {
4701d102b48SJeremy L Thompson     ierr = CeedVectorGetArray(impl->evecs[i+impl->numein], CEED_MEM_HOST,
471*e15f9bd0SJeremy L Thompson                               &impl->edata[i + numinputfields]); CeedChkBackend(ierr);
4721d102b48SJeremy L Thompson   }
4731d102b48SJeremy L Thompson 
4741d102b48SJeremy L Thompson   // Loop through elements
4751d102b48SJeremy L Thompson   for (CeedInt e=0; e<nblks*blksize; e+=blksize) {
4761d102b48SJeremy L Thompson     // Output pointers
4771d102b48SJeremy L Thompson     for (CeedInt i=0; i<numoutputfields; i++) {
4781d102b48SJeremy L Thompson       ierr = CeedQFunctionFieldGetEvalMode(qfoutputfields[i], &emode);
479*e15f9bd0SJeremy L Thompson       CeedChkBackend(ierr);
4801d102b48SJeremy L Thompson       if (emode == CEED_EVAL_NONE) {
4811d102b48SJeremy L Thompson         ierr = CeedQFunctionFieldGetSize(qfoutputfields[i], &size);
482*e15f9bd0SJeremy L Thompson         CeedChkBackend(ierr);
4831d102b48SJeremy L Thompson         ierr = CeedVectorSetArray(impl->qvecsout[i], CEED_MEM_HOST,
4841d102b48SJeremy L Thompson                                   CEED_USE_POINTER,
4851d102b48SJeremy L Thompson                                   &impl->edata[i + numinputfields][e*Q*size]);
486*e15f9bd0SJeremy L Thompson         CeedChkBackend(ierr);
4871d102b48SJeremy L Thompson       }
4881d102b48SJeremy L Thompson     }
4891d102b48SJeremy L Thompson 
49016911fdaSjeremylt     // Input basis apply
49116911fdaSjeremylt     ierr = CeedOperatorInputBasis_Blocked(e, Q, qfinputfields, opinputfields,
49216911fdaSjeremylt                                           numinputfields, blksize, false, impl);
493*e15f9bd0SJeremy L Thompson     CeedChkBackend(ierr);
49416911fdaSjeremylt 
4951d102b48SJeremy L Thompson     // Q function
49616911fdaSjeremylt     if (!impl->identityqf) {
4971d102b48SJeremy L Thompson       ierr = CeedQFunctionApply(qf, Q*blksize, impl->qvecsin, impl->qvecsout);
498*e15f9bd0SJeremy L Thompson       CeedChkBackend(ierr);
49916911fdaSjeremylt     }
5001d102b48SJeremy L Thompson 
5011d102b48SJeremy L Thompson     // Output basis apply
5021d102b48SJeremy L Thompson     ierr = CeedOperatorOutputBasis_Blocked(e, Q, qfoutputfields, opoutputfields,
5037f823360Sjeremylt                                            blksize, numinputfields,
5047f823360Sjeremylt                                            numoutputfields, op, impl);
505*e15f9bd0SJeremy L Thompson     CeedChkBackend(ierr);
5061d102b48SJeremy L Thompson   }
50789c6efa4Sjeremylt 
50889c6efa4Sjeremylt   // Output restriction
50989c6efa4Sjeremylt   for (CeedInt i=0; i<numoutputfields; i++) {
51089c6efa4Sjeremylt     // Restore evec
51189c6efa4Sjeremylt     ierr = CeedVectorRestoreArray(impl->evecs[i+impl->numein],
512*e15f9bd0SJeremy L Thompson                                   &impl->edata[i + numinputfields]); CeedChkBackend(ierr);
513d1bcdac9Sjeremylt     // Get output vector
514*e15f9bd0SJeremy L Thompson     ierr = CeedOperatorFieldGetVector(opoutputfields[i], &vec);
515*e15f9bd0SJeremy L Thompson     CeedChkBackend(ierr);
51689c6efa4Sjeremylt     // Active
517d1bcdac9Sjeremylt     if (vec == CEED_VECTOR_ACTIVE)
518d1bcdac9Sjeremylt       vec = outvec;
5194a2e7687Sjeremylt     // Restrict
520a8d32208Sjeremylt     ierr = CeedElemRestrictionApply(impl->blkrestr[i+impl->numein],
521a8d32208Sjeremylt                                     CEED_TRANSPOSE, impl->evecs[i+impl->numein],
522*e15f9bd0SJeremy L Thompson                                     vec, request); CeedChkBackend(ierr);
52389c6efa4Sjeremylt 
5244a2e7687Sjeremylt   }
5254a2e7687Sjeremylt 
5264a2e7687Sjeremylt   // Restore input arrays
5271d102b48SJeremy L Thompson   ierr = CeedOperatorRestoreInputs_Blocked(numinputfields, qfinputfields,
528*e15f9bd0SJeremy L Thompson          opinputfields, false, impl); CeedChkBackend(ierr);
5291d102b48SJeremy L Thompson 
530*e15f9bd0SJeremy L Thompson   return CEED_ERROR_SUCCESS;
5311d102b48SJeremy L Thompson }
5321d102b48SJeremy L Thompson 
533f10650afSjeremylt //------------------------------------------------------------------------------
5341d102b48SJeremy L Thompson // Assemble Linear QFunction
535f10650afSjeremylt //------------------------------------------------------------------------------
53680ac2e43SJeremy L Thompson static int CeedOperatorLinearAssembleQFunction_Blocked(CeedOperator op,
5371d102b48SJeremy L Thompson     CeedVector *assembled, CeedElemRestriction *rstr, CeedRequest *request) {
5381d102b48SJeremy L Thompson   int ierr;
5391d102b48SJeremy L Thompson   CeedOperator_Blocked *impl;
540*e15f9bd0SJeremy L Thompson   ierr = CeedOperatorGetData(op, &impl); CeedChkBackend(ierr);
5411d102b48SJeremy L Thompson   const CeedInt blksize = 8;
5421d102b48SJeremy L Thompson   CeedInt Q, numinputfields, numoutputfields, numelements, size;
543*e15f9bd0SJeremy L Thompson   ierr = CeedOperatorGetNumElements(op, &numelements); CeedChkBackend(ierr);
544*e15f9bd0SJeremy L Thompson   ierr = CeedOperatorGetNumQuadraturePoints(op, &Q); CeedChkBackend(ierr);
5451d102b48SJeremy L Thompson   CeedInt nblks = (numelements/blksize) + !!(numelements%blksize);
5461d102b48SJeremy L Thompson   CeedQFunction qf;
547*e15f9bd0SJeremy L Thompson   ierr = CeedOperatorGetQFunction(op, &qf); CeedChkBackend(ierr);
5481d102b48SJeremy L Thompson   ierr= CeedQFunctionGetNumArgs(qf, &numinputfields, &numoutputfields);
549*e15f9bd0SJeremy L Thompson   CeedChkBackend(ierr);
5501d102b48SJeremy L Thompson   CeedOperatorField *opinputfields, *opoutputfields;
5511d102b48SJeremy L Thompson   ierr = CeedOperatorGetFields(op, &opinputfields, &opoutputfields);
552*e15f9bd0SJeremy L Thompson   CeedChkBackend(ierr);
5531d102b48SJeremy L Thompson   CeedQFunctionField *qfinputfields, *qfoutputfields;
5541d102b48SJeremy L Thompson   ierr = CeedQFunctionGetFields(qf, &qfinputfields, &qfoutputfields);
555*e15f9bd0SJeremy L Thompson   CeedChkBackend(ierr);
5561d102b48SJeremy L Thompson   CeedVector vec, lvec;
5571d102b48SJeremy L Thompson   CeedInt numactivein = 0, numactiveout = 0;
55842ea3801Sjeremylt   CeedVector *activein = NULL;
5591d102b48SJeremy L Thompson   CeedScalar *a, *tmp;
5601d102b48SJeremy L Thompson   Ceed ceed;
561*e15f9bd0SJeremy L Thompson   ierr = CeedOperatorGetCeed(op, &ceed); CeedChkBackend(ierr);
5621d102b48SJeremy L Thompson 
5631d102b48SJeremy L Thompson   // Setup
564*e15f9bd0SJeremy L Thompson   ierr = CeedOperatorSetup_Blocked(op); CeedChkBackend(ierr);
5651d102b48SJeremy L Thompson 
56616911fdaSjeremylt   // Check for identity
56716911fdaSjeremylt   if (impl->identityqf)
56816911fdaSjeremylt     // LCOV_EXCL_START
569*e15f9bd0SJeremy L Thompson     return CeedError(ceed, CEED_ERROR_BACKEND,
570*e15f9bd0SJeremy L Thompson                      "Assembling identity qfunctions not supported");
57116911fdaSjeremylt   // LCOV_EXCL_STOP
57216911fdaSjeremylt 
5731d102b48SJeremy L Thompson   // Input Evecs and Restriction
5741d102b48SJeremy L Thompson   ierr = CeedOperatorSetupInputs_Blocked(numinputfields, qfinputfields,
5751d102b48SJeremy L Thompson                                          opinputfields, NULL, true, impl,
576*e15f9bd0SJeremy L Thompson                                          request); CeedChkBackend(ierr);
5771d102b48SJeremy L Thompson 
5781d102b48SJeremy L Thompson   // Count number of active input fields
5794a2e7687Sjeremylt   for (CeedInt i=0; i<numinputfields; i++) {
5801d102b48SJeremy L Thompson     // Get input vector
581*e15f9bd0SJeremy L Thompson     ierr = CeedOperatorFieldGetVector(opinputfields[i], &vec); CeedChkBackend(ierr);
5821d102b48SJeremy L Thompson     // Check if active input
5831d102b48SJeremy L Thompson     if (vec == CEED_VECTOR_ACTIVE) {
584*e15f9bd0SJeremy L Thompson       ierr = CeedQFunctionFieldGetSize(qfinputfields[i], &size); CeedChkBackend(ierr);
585*e15f9bd0SJeremy L Thompson       ierr = CeedVectorSetValue(impl->qvecsin[i], 0.0); CeedChkBackend(ierr);
5861d102b48SJeremy L Thompson       ierr = CeedVectorGetArray(impl->qvecsin[i], CEED_MEM_HOST, &tmp);
587*e15f9bd0SJeremy L Thompson       CeedChkBackend(ierr);
588*e15f9bd0SJeremy L Thompson       ierr = CeedRealloc(numactivein + size, &activein); CeedChkBackend(ierr);
5891d102b48SJeremy L Thompson       for (CeedInt field=0; field<size; field++) {
59042ea3801Sjeremylt         ierr = CeedVectorCreate(ceed, Q*blksize, &activein[numactivein+field]);
591*e15f9bd0SJeremy L Thompson         CeedChkBackend(ierr);
59242ea3801Sjeremylt         ierr = CeedVectorSetArray(activein[numactivein+field], CEED_MEM_HOST,
59342ea3801Sjeremylt                                   CEED_USE_POINTER, &tmp[field*Q*blksize]);
594*e15f9bd0SJeremy L Thompson         CeedChkBackend(ierr);
5951d102b48SJeremy L Thompson       }
5961d102b48SJeremy L Thompson       numactivein += size;
597*e15f9bd0SJeremy L Thompson       ierr = CeedVectorRestoreArray(impl->qvecsin[i], &tmp); CeedChkBackend(ierr);
5981d102b48SJeremy L Thompson     }
5991d102b48SJeremy L Thompson   }
6001d102b48SJeremy L Thompson 
6011d102b48SJeremy L Thompson   // Count number of active output fields
6021d102b48SJeremy L Thompson   for (CeedInt i=0; i<numoutputfields; i++) {
6031d102b48SJeremy L Thompson     // Get output vector
604*e15f9bd0SJeremy L Thompson     ierr = CeedOperatorFieldGetVector(opoutputfields[i], &vec);
605*e15f9bd0SJeremy L Thompson     CeedChkBackend(ierr);
6061d102b48SJeremy L Thompson     // Check if active output
6071d102b48SJeremy L Thompson     if (vec == CEED_VECTOR_ACTIVE) {
608*e15f9bd0SJeremy L Thompson       ierr = CeedQFunctionFieldGetSize(qfoutputfields[i], &size);
609*e15f9bd0SJeremy L Thompson       CeedChkBackend(ierr);
6101d102b48SJeremy L Thompson       numactiveout += size;
6111d102b48SJeremy L Thompson     }
6121d102b48SJeremy L Thompson   }
6131d102b48SJeremy L Thompson 
6141d102b48SJeremy L Thompson   // Check sizes
6151d102b48SJeremy L Thompson   if (!numactivein || !numactiveout)
6161d102b48SJeremy L Thompson     // LCOV_EXCL_START
617*e15f9bd0SJeremy L Thompson     return CeedError(ceed, CEED_ERROR_BACKEND,
618*e15f9bd0SJeremy L Thompson                      "Cannot assemble QFunction without active inputs "
6191d102b48SJeremy L Thompson                      "and outputs");
6201d102b48SJeremy L Thompson   // LCOV_EXCL_STOP
6211d102b48SJeremy L Thompson 
6221d102b48SJeremy L Thompson   // Setup lvec
6231d102b48SJeremy L Thompson   ierr = CeedVectorCreate(ceed, nblks*blksize*Q*numactivein*numactiveout,
624*e15f9bd0SJeremy L Thompson                           &lvec); CeedChkBackend(ierr);
625*e15f9bd0SJeremy L Thompson   ierr = CeedVectorGetArray(lvec, CEED_MEM_HOST, &a); CeedChkBackend(ierr);
6261d102b48SJeremy L Thompson 
6271d102b48SJeremy L Thompson   // Create output restriction
6287509a596Sjeremylt   CeedInt strides[3] = {1, Q, numactivein *numactiveout*Q};
629d979a051Sjeremylt   ierr = CeedElemRestrictionCreateStrided(ceed, numelements, Q,
630d979a051Sjeremylt                                           numactivein*numactiveout,
631d979a051Sjeremylt                                           numactivein*numactiveout*numelements*Q,
632*e15f9bd0SJeremy L Thompson                                           strides, rstr); CeedChkBackend(ierr);
6331d102b48SJeremy L Thompson   // Create assembled vector
6341d102b48SJeremy L Thompson   ierr = CeedVectorCreate(ceed, numelements*Q*numactivein*numactiveout,
635*e15f9bd0SJeremy L Thompson                           assembled); CeedChkBackend(ierr);
6361d102b48SJeremy L Thompson 
6371d102b48SJeremy L Thompson   // Loop through elements
6381d102b48SJeremy L Thompson   for (CeedInt e=0; e<nblks*blksize; e+=blksize) {
6391d102b48SJeremy L Thompson     // Input basis apply
6401d102b48SJeremy L Thompson     ierr = CeedOperatorInputBasis_Blocked(e, Q, qfinputfields, opinputfields,
6411d102b48SJeremy L Thompson                                           numinputfields, blksize, true, impl);
642*e15f9bd0SJeremy L Thompson     CeedChkBackend(ierr);
6431d102b48SJeremy L Thompson 
6441d102b48SJeremy L Thompson     // Assemble QFunction
6451d102b48SJeremy L Thompson     for (CeedInt in=0; in<numactivein; in++) {
6461d102b48SJeremy L Thompson       // Set Inputs
647*e15f9bd0SJeremy L Thompson       ierr = CeedVectorSetValue(activein[in], 1.0); CeedChkBackend(ierr);
64842ea3801Sjeremylt       if (numactivein > 1) {
64942ea3801Sjeremylt         ierr = CeedVectorSetValue(activein[(in+numactivein-1)%numactivein],
650*e15f9bd0SJeremy L Thompson                                   0.0); CeedChkBackend(ierr);
65142ea3801Sjeremylt       }
6521d102b48SJeremy L Thompson       // Set Outputs
6531d102b48SJeremy L Thompson       for (CeedInt out=0; out<numoutputfields; out++) {
6541d102b48SJeremy L Thompson         // Get output vector
6551d102b48SJeremy L Thompson         ierr = CeedOperatorFieldGetVector(opoutputfields[out], &vec);
656*e15f9bd0SJeremy L Thompson         CeedChkBackend(ierr);
6571d102b48SJeremy L Thompson         // Check if active output
6581d102b48SJeremy L Thompson         if (vec == CEED_VECTOR_ACTIVE) {
6591d102b48SJeremy L Thompson           CeedVectorSetArray(impl->qvecsout[out], CEED_MEM_HOST,
660*e15f9bd0SJeremy L Thompson                              CEED_USE_POINTER, a); CeedChkBackend(ierr);
6611d102b48SJeremy L Thompson           ierr = CeedQFunctionFieldGetSize(qfoutputfields[out], &size);
662*e15f9bd0SJeremy L Thompson           CeedChkBackend(ierr);
6631d102b48SJeremy L Thompson           a += size*Q*blksize; // Advance the pointer by the size of the output
6641d102b48SJeremy L Thompson         }
6651d102b48SJeremy L Thompson       }
6661d102b48SJeremy L Thompson       // Apply QFunction
6671d102b48SJeremy L Thompson       ierr = CeedQFunctionApply(qf, Q*blksize, impl->qvecsin, impl->qvecsout);
668*e15f9bd0SJeremy L Thompson       CeedChkBackend(ierr);
6694a2e7687Sjeremylt     }
6704a2e7687Sjeremylt   }
6714a2e7687Sjeremylt 
6721d102b48SJeremy L Thompson   // Un-set output Qvecs to prevent accidental overwrite of Assembled
6731d102b48SJeremy L Thompson   for (CeedInt out=0; out<numoutputfields; out++) {
6741d102b48SJeremy L Thompson     // Get output vector
6751d102b48SJeremy L Thompson     ierr = CeedOperatorFieldGetVector(opoutputfields[out], &vec);
676*e15f9bd0SJeremy L Thompson     CeedChkBackend(ierr);
6771d102b48SJeremy L Thompson     // Check if active output
6781d102b48SJeremy L Thompson     if (vec == CEED_VECTOR_ACTIVE) {
6791d102b48SJeremy L Thompson       CeedVectorSetArray(impl->qvecsout[out], CEED_MEM_HOST, CEED_COPY_VALUES,
680*e15f9bd0SJeremy L Thompson                          NULL); CeedChkBackend(ierr);
6811d102b48SJeremy L Thompson     }
6821d102b48SJeremy L Thompson   }
6831d102b48SJeremy L Thompson 
6841d102b48SJeremy L Thompson   // Restore input arrays
6851d102b48SJeremy L Thompson   ierr = CeedOperatorRestoreInputs_Blocked(numinputfields, qfinputfields,
686*e15f9bd0SJeremy L Thompson          opinputfields, true, impl); CeedChkBackend(ierr);
6871d102b48SJeremy L Thompson 
6881d102b48SJeremy L Thompson   // Output blocked restriction
689*e15f9bd0SJeremy L Thompson   ierr = CeedVectorRestoreArray(lvec, &a); CeedChkBackend(ierr);
690*e15f9bd0SJeremy L Thompson   ierr = CeedVectorSetValue(*assembled, 0.0); CeedChkBackend(ierr);
6911d102b48SJeremy L Thompson   CeedElemRestriction blkrstr;
6927509a596Sjeremylt   ierr = CeedElemRestrictionCreateBlockedStrided(ceed, numelements, Q, blksize,
693d979a051Sjeremylt          numactivein*numactiveout, numactivein*numactiveout*numelements*Q,
694*e15f9bd0SJeremy L Thompson          strides, &blkrstr); CeedChkBackend(ierr);
695a8d32208Sjeremylt   ierr = CeedElemRestrictionApply(blkrstr, CEED_TRANSPOSE, lvec, *assembled,
696*e15f9bd0SJeremy L Thompson                                   request); CeedChkBackend(ierr);
6971d102b48SJeremy L Thompson 
6981d102b48SJeremy L Thompson   // Cleanup
69942ea3801Sjeremylt   for (CeedInt i=0; i<numactivein; i++) {
700*e15f9bd0SJeremy L Thompson     ierr = CeedVectorDestroy(&activein[i]); CeedChkBackend(ierr);
70142ea3801Sjeremylt   }
702*e15f9bd0SJeremy L Thompson   ierr = CeedFree(&activein); CeedChkBackend(ierr);
703*e15f9bd0SJeremy L Thompson   ierr = CeedVectorDestroy(&lvec); CeedChkBackend(ierr);
704*e15f9bd0SJeremy L Thompson   ierr = CeedElemRestrictionDestroy(&blkrstr); CeedChkBackend(ierr);
7051d102b48SJeremy L Thompson 
706*e15f9bd0SJeremy L Thompson   return CEED_ERROR_SUCCESS;
7074a2e7687Sjeremylt }
7084a2e7687Sjeremylt 
709f10650afSjeremylt //------------------------------------------------------------------------------
710f10650afSjeremylt // Operator Destroy
711f10650afSjeremylt //------------------------------------------------------------------------------
712f10650afSjeremylt static int CeedOperatorDestroy_Blocked(CeedOperator op) {
713f10650afSjeremylt   int ierr;
714f10650afSjeremylt   CeedOperator_Blocked *impl;
715*e15f9bd0SJeremy L Thompson   ierr = CeedOperatorGetData(op, &impl); CeedChkBackend(ierr);
716f10650afSjeremylt 
717f10650afSjeremylt   for (CeedInt i=0; i<impl->numein+impl->numeout; i++) {
718*e15f9bd0SJeremy L Thompson     ierr = CeedElemRestrictionDestroy(&impl->blkrestr[i]); CeedChkBackend(ierr);
719*e15f9bd0SJeremy L Thompson     ierr = CeedVectorDestroy(&impl->evecs[i]); CeedChkBackend(ierr);
720f10650afSjeremylt   }
721*e15f9bd0SJeremy L Thompson   ierr = CeedFree(&impl->blkrestr); CeedChkBackend(ierr);
722*e15f9bd0SJeremy L Thompson   ierr = CeedFree(&impl->evecs); CeedChkBackend(ierr);
723*e15f9bd0SJeremy L Thompson   ierr = CeedFree(&impl->edata); CeedChkBackend(ierr);
724*e15f9bd0SJeremy L Thompson   ierr = CeedFree(&impl->inputstate); CeedChkBackend(ierr);
725f10650afSjeremylt 
726f10650afSjeremylt   for (CeedInt i=0; i<impl->numein; i++) {
727*e15f9bd0SJeremy L Thompson     ierr = CeedVectorDestroy(&impl->evecsin[i]); CeedChkBackend(ierr);
728*e15f9bd0SJeremy L Thompson     ierr = CeedVectorDestroy(&impl->qvecsin[i]); CeedChkBackend(ierr);
729f10650afSjeremylt   }
730*e15f9bd0SJeremy L Thompson   ierr = CeedFree(&impl->evecsin); CeedChkBackend(ierr);
731*e15f9bd0SJeremy L Thompson   ierr = CeedFree(&impl->qvecsin); CeedChkBackend(ierr);
732f10650afSjeremylt 
733f10650afSjeremylt   for (CeedInt i=0; i<impl->numeout; i++) {
734*e15f9bd0SJeremy L Thompson     ierr = CeedVectorDestroy(&impl->evecsout[i]); CeedChkBackend(ierr);
735*e15f9bd0SJeremy L Thompson     ierr = CeedVectorDestroy(&impl->qvecsout[i]); CeedChkBackend(ierr);
736f10650afSjeremylt   }
737*e15f9bd0SJeremy L Thompson   ierr = CeedFree(&impl->evecsout); CeedChkBackend(ierr);
738*e15f9bd0SJeremy L Thompson   ierr = CeedFree(&impl->qvecsout); CeedChkBackend(ierr);
739f10650afSjeremylt 
740*e15f9bd0SJeremy L Thompson   ierr = CeedFree(&impl); CeedChkBackend(ierr);
741*e15f9bd0SJeremy L Thompson   return CEED_ERROR_SUCCESS;
742f10650afSjeremylt }
743f10650afSjeremylt 
744f10650afSjeremylt //------------------------------------------------------------------------------
745f10650afSjeremylt // Operator Create
746f10650afSjeremylt //------------------------------------------------------------------------------
7474a2e7687Sjeremylt int CeedOperatorCreate_Blocked(CeedOperator op) {
7484a2e7687Sjeremylt   int ierr;
749fe2413ffSjeremylt   Ceed ceed;
750*e15f9bd0SJeremy L Thompson   ierr = CeedOperatorGetCeed(op, &ceed); CeedChkBackend(ierr);
7514ce2993fSjeremylt   CeedOperator_Blocked *impl;
7524a2e7687Sjeremylt 
753*e15f9bd0SJeremy L Thompson   ierr = CeedCalloc(1, &impl); CeedChkBackend(ierr);
754*e15f9bd0SJeremy L Thompson   ierr = CeedOperatorSetData(op, impl); CeedChkBackend(ierr);
755fe2413ffSjeremylt 
75680ac2e43SJeremy L Thompson   ierr = CeedSetBackendFunction(ceed, "Operator", op, "LinearAssembleQFunction",
75780ac2e43SJeremy L Thompson                                 CeedOperatorLinearAssembleQFunction_Blocked);
758*e15f9bd0SJeremy L Thompson   CeedChkBackend(ierr);
759cae8b89aSjeremylt   ierr = CeedSetBackendFunction(ceed, "Operator", op, "ApplyAdd",
760*e15f9bd0SJeremy L Thompson                                 CeedOperatorApplyAdd_Blocked); CeedChkBackend(ierr);
761fe2413ffSjeremylt   ierr = CeedSetBackendFunction(ceed, "Operator", op, "Destroy",
762*e15f9bd0SJeremy L Thompson                                 CeedOperatorDestroy_Blocked); CeedChkBackend(ierr);
763*e15f9bd0SJeremy L Thompson   return CEED_ERROR_SUCCESS;
7644a2e7687Sjeremylt }
765f10650afSjeremylt //------------------------------------------------------------------------------
766