xref: /libCEED/rust/libceed-sys/c-src/backends/blocked/ceed-blocked-operator.c (revision 4ce2993fee6e7bc9b86526ee90098d0dc489fc60)
14a2e7687Sjeremylt // Copyright (c) 2017-2018, Lawrence Livermore National Security, LLC.
24a2e7687Sjeremylt // Produced at the Lawrence Livermore National Laboratory. LLNL-CODE-734707.
34a2e7687Sjeremylt // All Rights reserved. See files LICENSE and NOTICE for details.
44a2e7687Sjeremylt //
54a2e7687Sjeremylt // This file is part of CEED, a collection of benchmarks, miniapps, software
64a2e7687Sjeremylt // libraries and APIs for efficient high-order finite element and spectral
74a2e7687Sjeremylt // element discretizations for exascale applications. For more information and
84a2e7687Sjeremylt // source code availability see http://github.com/ceed.
94a2e7687Sjeremylt //
104a2e7687Sjeremylt // The CEED research is supported by the Exascale Computing Project 17-SC-20-SC,
114a2e7687Sjeremylt // a collaborative effort of two U.S. Department of Energy organizations (Office
124a2e7687Sjeremylt // of Science and the National Nuclear Security Administration) responsible for
134a2e7687Sjeremylt // the planning and preparation of a capable exascale ecosystem, including
144a2e7687Sjeremylt // software, applications, hardware, advanced system engineering and early
154a2e7687Sjeremylt // testbed platforms, in support of the nation's exascale computing imperative.
164a2e7687Sjeremylt 
174a2e7687Sjeremylt #include <ceed-impl.h>
184a2e7687Sjeremylt #include <string.h>
194a2e7687Sjeremylt #include "ceed-blocked.h"
204a2e7687Sjeremylt #include "../ref/ceed-ref.h"
214a2e7687Sjeremylt 
224a2e7687Sjeremylt static int CeedOperatorDestroy_Blocked(CeedOperator op) {
234a2e7687Sjeremylt   int ierr;
24*4ce2993fSjeremylt   CeedOperator_Blocked *impl;
25*4ce2993fSjeremylt   ierr = CeedOperatorGetData(op, (void*)&impl); CeedChk(ierr);
264a2e7687Sjeremylt 
274a2e7687Sjeremylt   for (CeedInt i=0; i<impl->numein+impl->numeout; i++) {
284a2e7687Sjeremylt     ierr = CeedElemRestrictionDestroy(&impl->blkrestr[i]); CeedChk(ierr);
294a2e7687Sjeremylt     ierr = CeedVectorDestroy(&impl->evecs[i]); CeedChk(ierr);
304a2e7687Sjeremylt   }
314a2e7687Sjeremylt   ierr = CeedFree(&impl->blkrestr); CeedChk(ierr);
324a2e7687Sjeremylt   ierr = CeedFree(&impl->evecs); CeedChk(ierr);
334a2e7687Sjeremylt   ierr = CeedFree(&impl->edata); CeedChk(ierr);
344a2e7687Sjeremylt 
354a2e7687Sjeremylt   for (CeedInt i=0; i<impl->numqin+impl->numqout; i++) {
364a2e7687Sjeremylt     ierr = CeedFree(&impl->qdata_alloc[i]); CeedChk(ierr);
374a2e7687Sjeremylt   }
384a2e7687Sjeremylt   ierr = CeedFree(&impl->qdata_alloc); CeedChk(ierr);
394a2e7687Sjeremylt   ierr = CeedFree(&impl->qdata); CeedChk(ierr);
404a2e7687Sjeremylt 
414a2e7687Sjeremylt   ierr = CeedFree(&impl->indata); CeedChk(ierr);
424a2e7687Sjeremylt   ierr = CeedFree(&impl->outdata); CeedChk(ierr);
434a2e7687Sjeremylt 
444a2e7687Sjeremylt   ierr = CeedFree(&op->data); CeedChk(ierr);
454a2e7687Sjeremylt   return 0;
464a2e7687Sjeremylt }
474a2e7687Sjeremylt 
484a2e7687Sjeremylt /*
494a2e7687Sjeremylt   Setup infields or outfields
504a2e7687Sjeremylt  */
514a2e7687Sjeremylt static int CeedOperatorSetupFields_Blocked(struct CeedQFunctionField qfields[16],
524a2e7687Sjeremylt                                        struct CeedOperatorField ofields[16],
534a2e7687Sjeremylt                                        CeedElemRestriction *blkrestr,
544a2e7687Sjeremylt                                        CeedVector *evecs, CeedScalar **qdata,
554a2e7687Sjeremylt                                        CeedScalar **qdata_alloc, CeedScalar **indata,
564a2e7687Sjeremylt                                        CeedInt starti, CeedInt startq,
574a2e7687Sjeremylt                                        CeedInt numfields, CeedInt Q) {
584a2e7687Sjeremylt   CeedInt dim, ierr, iq=startq, ncomp;
594a2e7687Sjeremylt   const CeedInt blksize = 8;
604a2e7687Sjeremylt 
614a2e7687Sjeremylt   // Loop over fields
624a2e7687Sjeremylt   for (CeedInt i=0; i<numfields; i++) {
634a2e7687Sjeremylt     CeedEvalMode emode = qfields[i].emode;
644a2e7687Sjeremylt 
654a2e7687Sjeremylt     if (emode != CEED_EVAL_WEIGHT) {
664a2e7687Sjeremylt       CeedElemRestriction r = ofields[i].Erestrict;
674a2e7687Sjeremylt       CeedElemRestriction_Ref *data = r->data;
68*4ce2993fSjeremylt       Ceed ceed;
69*4ce2993fSjeremylt       ierr = CeedElemRestrictionGetCeed(r, &ceed); CeedChk(ierr);
70*4ce2993fSjeremylt       CeedInt nelem, elemsize, ndof, ncomp;
71*4ce2993fSjeremylt       ierr = CeedElemRestrictionGetNumElements(r, &nelem); CeedChk(ierr);
72*4ce2993fSjeremylt       ierr = CeedElemRestrictionGetElementSize(r, &elemsize); CeedChk(ierr);
73*4ce2993fSjeremylt       ierr = CeedElemRestrictionGetNumDoF(r, &ndof); CeedChk(ierr);
74*4ce2993fSjeremylt       ierr = CeedElemRestrictionGetNumComponents(r, &ncomp); CeedChk(ierr);
75*4ce2993fSjeremylt       ierr = CeedElemRestrictionCreateBlocked(ceed, nelem, elemsize,
76*4ce2993fSjeremylt                                               blksize, ndof, ncomp,
774a2e7687Sjeremylt                                               CEED_MEM_HOST, CEED_COPY_VALUES,
784a2e7687Sjeremylt                                               data->indices, &blkrestr[i+starti]);
794a2e7687Sjeremylt       CeedChk(ierr);
804a2e7687Sjeremylt       ierr = CeedElemRestrictionCreateVector(blkrestr[i+starti], NULL,
814a2e7687Sjeremylt                                              &evecs[i+starti]);
824a2e7687Sjeremylt       CeedChk(ierr);
834a2e7687Sjeremylt     }
844a2e7687Sjeremylt 
854a2e7687Sjeremylt     switch(emode) {
864a2e7687Sjeremylt     case CEED_EVAL_NONE:
874a2e7687Sjeremylt       break; // No action
884a2e7687Sjeremylt     case CEED_EVAL_INTERP:
894a2e7687Sjeremylt       ncomp = qfields[i].ncomp;
904a2e7687Sjeremylt       ierr = CeedMalloc(Q*ncomp*blksize, &qdata_alloc[iq]); CeedChk(ierr);
914a2e7687Sjeremylt       qdata[i + starti] = qdata_alloc[iq];
924a2e7687Sjeremylt       iq++;
934a2e7687Sjeremylt       break;
944a2e7687Sjeremylt     case CEED_EVAL_GRAD:
954a2e7687Sjeremylt       ncomp = qfields[i].ncomp;
96*4ce2993fSjeremylt       ierr = CeedBasisGetDimension(ofields[i].basis, &dim); CeedChk(ierr);
974a2e7687Sjeremylt       ierr = CeedMalloc(Q*ncomp*dim*blksize, &qdata_alloc[iq]); CeedChk(ierr);
984a2e7687Sjeremylt       qdata[i + starti] = qdata_alloc[iq];
994a2e7687Sjeremylt       iq++;
1004a2e7687Sjeremylt       break;
1014a2e7687Sjeremylt     case CEED_EVAL_WEIGHT: // Only on input fields
1024a2e7687Sjeremylt       ierr = CeedMalloc(Q*blksize, &qdata_alloc[iq]); CeedChk(ierr);
1034a2e7687Sjeremylt       ierr = CeedBasisApply(ofields[iq].basis, blksize, CEED_NOTRANSPOSE,
1044a2e7687Sjeremylt                             CEED_EVAL_WEIGHT, NULL, qdata_alloc[iq]); CeedChk(ierr);
1054a2e7687Sjeremylt       qdata[i] = qdata_alloc[iq];
1064a2e7687Sjeremylt       indata[i] = qdata[i];
1074a2e7687Sjeremylt       iq++;
1084a2e7687Sjeremylt       break;
1094a2e7687Sjeremylt     case CEED_EVAL_DIV:
1104a2e7687Sjeremylt       break; // Not implimented
1114a2e7687Sjeremylt     case CEED_EVAL_CURL:
1124a2e7687Sjeremylt       break; // Not implimented
1134a2e7687Sjeremylt     }
1144a2e7687Sjeremylt   }
1154a2e7687Sjeremylt   return 0;
1164a2e7687Sjeremylt }
1174a2e7687Sjeremylt 
1184a2e7687Sjeremylt /*
1194a2e7687Sjeremylt   CeedOperator needs to connect all the named fields (be they active or passive)
1204a2e7687Sjeremylt   to the named inputs and outputs of its CeedQFunction.
1214a2e7687Sjeremylt  */
1224a2e7687Sjeremylt static int CeedOperatorSetup_Blocked(CeedOperator op) {
1234a2e7687Sjeremylt   int ierr;
124*4ce2993fSjeremylt   bool setupdone;
125*4ce2993fSjeremylt   ierr = CeedOperatorGetSetupStatus(op, &setupdone); CeedChk(ierr);
126*4ce2993fSjeremylt   if (setupdone) return 0;
127*4ce2993fSjeremylt   CeedOperator_Blocked *impl;
128*4ce2993fSjeremylt   ierr = CeedOperatorGetData(op, (void*)&impl); CeedChk(ierr);
129*4ce2993fSjeremylt   CeedQFunction qf;
130*4ce2993fSjeremylt   ierr = CeedOperatorGetQFunction(op, &qf); CeedChk(ierr);
131*4ce2993fSjeremylt   CeedInt Q, numinputfields, numoutputfields;
132*4ce2993fSjeremylt   ierr = CeedOperatorGetNumQuadraturePoints(op, &Q); CeedChk(ierr);
1334a2e7687Sjeremylt   ierr= CeedQFunctionGetNumArgs(qf, &numinputfields, &numoutputfields);
1344a2e7687Sjeremylt   CeedChk(ierr);
135*4ce2993fSjeremylt 
136*4ce2993fSjeremylt   // Count infield and outfield array sizes and evectors
1374a2e7687Sjeremylt   impl->numein = numinputfields;
1384a2e7687Sjeremylt   for (CeedInt i=0; i<numinputfields; i++) {
1394a2e7687Sjeremylt     CeedEvalMode emode = qf->inputfields[i].emode;
1404a2e7687Sjeremylt     impl->numqin += !!(emode & CEED_EVAL_INTERP) + !!(emode & CEED_EVAL_GRAD) +
1414a2e7687Sjeremylt                     !!(emode & CEED_EVAL_WEIGHT);
1424a2e7687Sjeremylt   }
1434a2e7687Sjeremylt   impl->numeout = numoutputfields;
1444a2e7687Sjeremylt   for (CeedInt i=0; i<numoutputfields; i++) {
1454a2e7687Sjeremylt     CeedEvalMode emode = qf->outputfields[i].emode;
1464a2e7687Sjeremylt     impl->numqout += !!(emode & CEED_EVAL_INTERP) + !!(emode & CEED_EVAL_GRAD);
1474a2e7687Sjeremylt   }
1484a2e7687Sjeremylt 
1494a2e7687Sjeremylt   // Allocate
1504a2e7687Sjeremylt   ierr = CeedCalloc(impl->numein + impl->numeout, &impl->blkrestr);
1514a2e7687Sjeremylt   CeedChk(ierr);
1524a2e7687Sjeremylt   ierr = CeedCalloc(impl->numein + impl->numeout, &impl->evecs);
1534a2e7687Sjeremylt   CeedChk(ierr);
1544a2e7687Sjeremylt   ierr = CeedCalloc(impl->numein + impl->numeout, &impl->edata);
1554a2e7687Sjeremylt   CeedChk(ierr);
1564a2e7687Sjeremylt 
1574a2e7687Sjeremylt   ierr = CeedCalloc(impl->numqin + impl->numqout, &impl->qdata_alloc);
1584a2e7687Sjeremylt   CeedChk(ierr);
1594a2e7687Sjeremylt   ierr = CeedCalloc(numinputfields + numoutputfields, &impl->qdata);
1604a2e7687Sjeremylt   CeedChk(ierr);
1614a2e7687Sjeremylt 
1624a2e7687Sjeremylt   ierr = CeedCalloc(16, &impl->indata); CeedChk(ierr);
1634a2e7687Sjeremylt   ierr = CeedCalloc(16, &impl->outdata); CeedChk(ierr);
1644a2e7687Sjeremylt   // Set up infield and outfield pointer arrays
1654a2e7687Sjeremylt   // Infields
1664a2e7687Sjeremylt   ierr = CeedOperatorSetupFields_Blocked(qf->inputfields, op->inputfields,
1674a2e7687Sjeremylt                                      impl->blkrestr, impl->evecs,
1684a2e7687Sjeremylt                                      impl->qdata, impl->qdata_alloc,
1694a2e7687Sjeremylt                                      impl->indata, 0,
1704a2e7687Sjeremylt                                      0, numinputfields, Q);
1714a2e7687Sjeremylt   CeedChk(ierr);
1724a2e7687Sjeremylt   // Outfields
1734a2e7687Sjeremylt   ierr = CeedOperatorSetupFields_Blocked(qf->outputfields, op->outputfields,
1744a2e7687Sjeremylt                                      impl->blkrestr, impl->evecs,
1754a2e7687Sjeremylt                                      impl->qdata, impl->qdata_alloc,
1764a2e7687Sjeremylt                                      impl->indata, numinputfields,
1774a2e7687Sjeremylt                                      impl->numqin, numoutputfields, Q);
1784a2e7687Sjeremylt   CeedChk(ierr);
1794a2e7687Sjeremylt   // Input Qvecs
1804a2e7687Sjeremylt   for (CeedInt i=0; i<numinputfields; i++) {
1814a2e7687Sjeremylt     CeedEvalMode emode = qf->inputfields[i].emode;
1824a2e7687Sjeremylt     if ((emode != CEED_EVAL_NONE) && (emode != CEED_EVAL_WEIGHT))
1834a2e7687Sjeremylt       impl->indata[i] =  impl->qdata[i];
1844a2e7687Sjeremylt   }
1854a2e7687Sjeremylt   // Output Qvecs
1864a2e7687Sjeremylt   for (CeedInt i=0; i<numoutputfields; i++) {
1874a2e7687Sjeremylt     CeedEvalMode emode = qf->outputfields[i].emode;
1884a2e7687Sjeremylt     if (emode != CEED_EVAL_NONE)
1894a2e7687Sjeremylt       impl->outdata[i] =  impl->qdata[i + numinputfields];
1904a2e7687Sjeremylt   }
1914a2e7687Sjeremylt 
192*4ce2993fSjeremylt   ierr = CeedOperatorSetSetupDone(op); CeedChk(ierr);
1934a2e7687Sjeremylt 
1944a2e7687Sjeremylt   return 0;
1954a2e7687Sjeremylt }
1964a2e7687Sjeremylt 
1974a2e7687Sjeremylt static int CeedOperatorApply_Blocked(CeedOperator op, CeedVector invec,
1984a2e7687Sjeremylt                                  CeedVector outvec, CeedRequest *request) {
1994a2e7687Sjeremylt   int ierr;
200*4ce2993fSjeremylt   CeedOperator_Blocked *impl;
201*4ce2993fSjeremylt   ierr = CeedOperatorGetData(op, (void*)&impl); CeedChk(ierr);
202*4ce2993fSjeremylt   const CeedInt blksize = 8;
203*4ce2993fSjeremylt   CeedInt Q, elemsize, numinputfields, numoutputfields, numelements;
204*4ce2993fSjeremylt   ierr = CeedOperatorGetNumElements(op, &numelements); CeedChk(ierr);
205*4ce2993fSjeremylt   ierr = CeedOperatorGetNumQuadraturePoints(op, &Q); CeedChk(ierr);
206*4ce2993fSjeremylt   CeedInt nblks = (numelements/blksize) + !!(numelements%blksize);
207*4ce2993fSjeremylt   CeedQFunction qf;
208*4ce2993fSjeremylt   ierr = CeedOperatorGetQFunction(op, &qf); CeedChk(ierr);
209*4ce2993fSjeremylt   ierr= CeedQFunctionGetNumArgs(qf, &numinputfields, &numoutputfields);
210*4ce2993fSjeremylt   CeedChk(ierr);
2114a2e7687Sjeremylt   CeedTransposeMode lmode = CEED_NOTRANSPOSE;
2124a2e7687Sjeremylt 
2134a2e7687Sjeremylt   // Setup
2144a2e7687Sjeremylt   ierr = CeedOperatorSetup_Blocked(op); CeedChk(ierr);
2154a2e7687Sjeremylt 
2164a2e7687Sjeremylt   // Input Evecs and Restriction
2174a2e7687Sjeremylt   for (CeedInt i=0; i<numinputfields; i++) {
2184a2e7687Sjeremylt     CeedEvalMode emode = qf->inputfields[i].emode;
2194a2e7687Sjeremylt     if (emode == CEED_EVAL_WEIGHT) { // Skip
2204a2e7687Sjeremylt     } else {
2214a2e7687Sjeremylt       // Active
2224a2e7687Sjeremylt       // Restrict
2234a2e7687Sjeremylt       if (op->inputfields[i].vec == CEED_VECTOR_ACTIVE) {
2244a2e7687Sjeremylt         ierr = CeedElemRestrictionApply(impl->blkrestr[i], CEED_NOTRANSPOSE,
2254a2e7687Sjeremylt                                         lmode, invec, impl->evecs[i],
2264a2e7687Sjeremylt                                         request); CeedChk(ierr); CeedChk(ierr);
2274a2e7687Sjeremylt       } else {
2284a2e7687Sjeremylt         // Passive
2294a2e7687Sjeremylt         // Restrict
2304a2e7687Sjeremylt         ierr = CeedElemRestrictionApply(impl->blkrestr[i], CEED_NOTRANSPOSE,
2314a2e7687Sjeremylt                                         lmode, op->inputfields[i].vec, impl->evecs[i],
2324a2e7687Sjeremylt                                         request); CeedChk(ierr);
2334a2e7687Sjeremylt       }
2344a2e7687Sjeremylt       // Get evec
2354a2e7687Sjeremylt       ierr = CeedVectorGetArrayRead(impl->evecs[i], CEED_MEM_HOST,
2364a2e7687Sjeremylt                                     (const CeedScalar **) &impl->edata[i]);
2374a2e7687Sjeremylt       CeedChk(ierr);
2384a2e7687Sjeremylt     }
2394a2e7687Sjeremylt   }
2404a2e7687Sjeremylt 
2414a2e7687Sjeremylt   // Output Evecs
2424a2e7687Sjeremylt   for (CeedInt i=0; i<numoutputfields; i++) {
2434a2e7687Sjeremylt     ierr = CeedVectorGetArray(impl->evecs[i+impl->numein], CEED_MEM_HOST,
2444a2e7687Sjeremylt                               &impl->edata[i + numinputfields]); CeedChk(ierr);
2454a2e7687Sjeremylt   }
2464a2e7687Sjeremylt 
2474a2e7687Sjeremylt   // Loop through elements
2484a2e7687Sjeremylt   for (CeedInt e=0; e<nblks*blksize; e+=blksize) {
2494a2e7687Sjeremylt     // Input basis apply if needed
2504a2e7687Sjeremylt     for (CeedInt i=0; i<numinputfields; i++) {
2514a2e7687Sjeremylt       // Get elemsize, emode, ncomp
252*4ce2993fSjeremylt       ierr = CeedElemRestrictionGetElementSize(
253*4ce2993fSjeremylt                op->inputfields[i].Erestrict, &elemsize); CeedChk(ierr);
2544a2e7687Sjeremylt       CeedEvalMode emode = qf->inputfields[i].emode;
2554a2e7687Sjeremylt       CeedInt ncomp = qf->inputfields[i].ncomp;
2564a2e7687Sjeremylt       // Basis action
2574a2e7687Sjeremylt       switch(emode) {
2584a2e7687Sjeremylt       case CEED_EVAL_NONE:
2594a2e7687Sjeremylt         impl->indata[i] = &impl->edata[i][e*Q*ncomp];
2604a2e7687Sjeremylt         break;
2614a2e7687Sjeremylt       case CEED_EVAL_INTERP:
2624a2e7687Sjeremylt         ierr = CeedBasisApply(op->inputfields[i].basis, blksize, CEED_NOTRANSPOSE,
2634a2e7687Sjeremylt                               CEED_EVAL_INTERP, &impl->edata[i][e*elemsize*ncomp], impl->qdata[i]);
2644a2e7687Sjeremylt         CeedChk(ierr);
2654a2e7687Sjeremylt         break;
2664a2e7687Sjeremylt       case CEED_EVAL_GRAD:
2674a2e7687Sjeremylt         ierr = CeedBasisApply(op->inputfields[i].basis, blksize, CEED_NOTRANSPOSE,
2684a2e7687Sjeremylt                               CEED_EVAL_GRAD, &impl->edata[i][e*elemsize*ncomp], impl->qdata[i]);
2694a2e7687Sjeremylt         CeedChk(ierr);
2704a2e7687Sjeremylt         break;
2714a2e7687Sjeremylt       case CEED_EVAL_WEIGHT:
2724a2e7687Sjeremylt         break;  // No action
2734a2e7687Sjeremylt       case CEED_EVAL_DIV:
2744a2e7687Sjeremylt         break; // Not implimented
2754a2e7687Sjeremylt       case CEED_EVAL_CURL:
2764a2e7687Sjeremylt         break; // Not implimented
2774a2e7687Sjeremylt       }
2784a2e7687Sjeremylt     }
2794a2e7687Sjeremylt 
2804a2e7687Sjeremylt     // Output pointers
2814a2e7687Sjeremylt     for (CeedInt i=0; i<numoutputfields; i++) {
2824a2e7687Sjeremylt       CeedEvalMode emode = qf->outputfields[i].emode;
2834a2e7687Sjeremylt       if (emode == CEED_EVAL_NONE) {
2844a2e7687Sjeremylt         CeedInt ncomp = qf->outputfields[i].ncomp;
2854a2e7687Sjeremylt         impl->outdata[i] = &impl->edata[i + numinputfields][e*Q*ncomp];
2864a2e7687Sjeremylt       }
2874a2e7687Sjeremylt     }
2884a2e7687Sjeremylt     // Q function
289*4ce2993fSjeremylt     ierr = CeedQFunctionApply(qf, Q*blksize,
2904a2e7687Sjeremylt                               (const CeedScalar * const*) impl->indata,
2914a2e7687Sjeremylt                               impl->outdata); CeedChk(ierr);
2924a2e7687Sjeremylt 
2934a2e7687Sjeremylt     // Output basis apply if needed
2944a2e7687Sjeremylt     for (CeedInt i=0; i<numoutputfields; i++) {
2954a2e7687Sjeremylt       // Get elemsize, emode, ncomp
296*4ce2993fSjeremylt       ierr = CeedElemRestrictionGetElementSize(
297*4ce2993fSjeremylt                op->outputfields[i].Erestrict, &elemsize); CeedChk(ierr);
2984a2e7687Sjeremylt       CeedInt ncomp = qf->outputfields[i].ncomp;
2994a2e7687Sjeremylt       CeedEvalMode emode = qf->outputfields[i].emode;
3004a2e7687Sjeremylt       // Basis action
3014a2e7687Sjeremylt       switch(emode) {
3024a2e7687Sjeremylt       case CEED_EVAL_NONE:
3034a2e7687Sjeremylt         break; // No action
3044a2e7687Sjeremylt       case CEED_EVAL_INTERP:
3054a2e7687Sjeremylt         ierr = CeedBasisApply(op->outputfields[i].basis, blksize, CEED_TRANSPOSE,
3064a2e7687Sjeremylt                               CEED_EVAL_INTERP, impl->outdata[i],
3074a2e7687Sjeremylt                               &impl->edata[i + numinputfields][e*elemsize*ncomp]);
3084a2e7687Sjeremylt         CeedChk(ierr);
3094a2e7687Sjeremylt         break;
3104a2e7687Sjeremylt       case CEED_EVAL_GRAD:
3114a2e7687Sjeremylt         ierr = CeedBasisApply(op->outputfields[i].basis, blksize, CEED_TRANSPOSE,
3124a2e7687Sjeremylt                               CEED_EVAL_GRAD,
3134a2e7687Sjeremylt                               impl->outdata[i], &impl->edata[i + numinputfields][e*elemsize*ncomp]);
3144a2e7687Sjeremylt         CeedChk(ierr);
3154a2e7687Sjeremylt         break;
316*4ce2993fSjeremylt       case CEED_EVAL_WEIGHT: {
317*4ce2993fSjeremylt         Ceed ceed;
318*4ce2993fSjeremylt         ierr = CeedOperatorGetCeed(op, &ceed); CeedChk(ierr);
319*4ce2993fSjeremylt         return CeedError(ceed, 1,
3204a2e7687Sjeremylt                          "CEED_EVAL_WEIGHT cannot be an output evaluation mode");
3214a2e7687Sjeremylt         break; // Should not occur
322*4ce2993fSjeremylt       }
3234a2e7687Sjeremylt       case CEED_EVAL_DIV:
3244a2e7687Sjeremylt         break; // Not implimented
3254a2e7687Sjeremylt       case CEED_EVAL_CURL:
3264a2e7687Sjeremylt         break; // Not implimented
3274a2e7687Sjeremylt       }
3284a2e7687Sjeremylt     }
3294a2e7687Sjeremylt   }
3304a2e7687Sjeremylt 
3314a2e7687Sjeremylt   // Zero lvecs
3324a2e7687Sjeremylt   ierr = CeedVectorSetValue(outvec, 0.0); CeedChk(ierr);
333*4ce2993fSjeremylt   for (CeedInt i=0; i<numoutputfields; i++)
3344a2e7687Sjeremylt     if (op->outputfields[i].vec != CEED_VECTOR_ACTIVE) {
3354a2e7687Sjeremylt       ierr = CeedVectorSetValue(op->outputfields[i].vec, 0.0); CeedChk(ierr);
3364a2e7687Sjeremylt     }
3374a2e7687Sjeremylt 
3384a2e7687Sjeremylt   // Output restriction
3394a2e7687Sjeremylt   for (CeedInt i=0; i<numoutputfields; i++) {
3404a2e7687Sjeremylt     // Restore evec
3414a2e7687Sjeremylt     ierr = CeedVectorRestoreArray(impl->evecs[i+impl->numein],
3424a2e7687Sjeremylt                                   &impl->edata[i + numinputfields]); CeedChk(ierr);
3434a2e7687Sjeremylt     // Active
3444a2e7687Sjeremylt     if (op->outputfields[i].vec == CEED_VECTOR_ACTIVE) {
3454a2e7687Sjeremylt       // Restrict
3464a2e7687Sjeremylt       ierr = CeedElemRestrictionApply(impl->blkrestr[i+impl->numein], CEED_TRANSPOSE,
3474a2e7687Sjeremylt                                       lmode, impl->evecs[i+impl->numein], outvec, request); CeedChk(ierr);
3484a2e7687Sjeremylt     } else {
3494a2e7687Sjeremylt       // Passive
3504a2e7687Sjeremylt       // Restrict
3514a2e7687Sjeremylt       ierr = CeedElemRestrictionApply(impl->blkrestr[i+impl->numein], CEED_TRANSPOSE,
3524a2e7687Sjeremylt                                       lmode, impl->evecs[i+impl->numein], op->outputfields[i].vec,
3534a2e7687Sjeremylt                                       request); CeedChk(ierr);
3544a2e7687Sjeremylt     }
3554a2e7687Sjeremylt   }
3564a2e7687Sjeremylt 
3574a2e7687Sjeremylt   // Restore input arrays
3584a2e7687Sjeremylt   for (CeedInt i=0; i<numinputfields; i++) {
3594a2e7687Sjeremylt     CeedEvalMode emode = qf->inputfields[i].emode;
3604a2e7687Sjeremylt     if (emode == CEED_EVAL_WEIGHT) { // Skip
3614a2e7687Sjeremylt     } else {
3624a2e7687Sjeremylt       ierr = CeedVectorRestoreArrayRead(impl->evecs[i],
3634a2e7687Sjeremylt                                         (const CeedScalar **) &impl->edata[i]); CeedChk(ierr);
3644a2e7687Sjeremylt     }
3654a2e7687Sjeremylt   }
3664a2e7687Sjeremylt 
3674a2e7687Sjeremylt   return 0;
3684a2e7687Sjeremylt }
3694a2e7687Sjeremylt 
3704a2e7687Sjeremylt int CeedOperatorCreate_Blocked(CeedOperator op) {
3714a2e7687Sjeremylt   int ierr;
372*4ce2993fSjeremylt   CeedOperator_Blocked *impl;
3734a2e7687Sjeremylt 
3744a2e7687Sjeremylt   ierr = CeedCalloc(1, &impl); CeedChk(ierr);
3754a2e7687Sjeremylt   op->data = impl;
3764a2e7687Sjeremylt   op->Destroy = CeedOperatorDestroy_Blocked;
3774a2e7687Sjeremylt   op->Apply = CeedOperatorApply_Blocked;
3784a2e7687Sjeremylt   return 0;
3794a2e7687Sjeremylt }
380