14a2e7687Sjeremylt // Copyright (c) 2017-2018, Lawrence Livermore National Security, LLC. 24a2e7687Sjeremylt // Produced at the Lawrence Livermore National Laboratory. LLNL-CODE-734707. 34a2e7687Sjeremylt // All Rights reserved. See files LICENSE and NOTICE for details. 44a2e7687Sjeremylt // 54a2e7687Sjeremylt // This file is part of CEED, a collection of benchmarks, miniapps, software 64a2e7687Sjeremylt // libraries and APIs for efficient high-order finite element and spectral 74a2e7687Sjeremylt // element discretizations for exascale applications. For more information and 84a2e7687Sjeremylt // source code availability see http://github.com/ceed. 94a2e7687Sjeremylt // 104a2e7687Sjeremylt // The CEED research is supported by the Exascale Computing Project 17-SC-20-SC, 114a2e7687Sjeremylt // a collaborative effort of two U.S. Department of Energy organizations (Office 124a2e7687Sjeremylt // of Science and the National Nuclear Security Administration) responsible for 134a2e7687Sjeremylt // the planning and preparation of a capable exascale ecosystem, including 144a2e7687Sjeremylt // software, applications, hardware, advanced system engineering and early 154a2e7687Sjeremylt // testbed platforms, in support of the nation's exascale computing imperative. 164a2e7687Sjeremylt 174a2e7687Sjeremylt #include <string.h> 184a2e7687Sjeremylt #include "ceed-blocked.h" 194a2e7687Sjeremylt #include "../ref/ceed-ref.h" 204a2e7687Sjeremylt 214a2e7687Sjeremylt static int CeedOperatorDestroy_Blocked(CeedOperator op) { 224a2e7687Sjeremylt int ierr; 234ce2993fSjeremylt CeedOperator_Blocked *impl; 244ce2993fSjeremylt ierr = CeedOperatorGetData(op, (void*)&impl); CeedChk(ierr); 254a2e7687Sjeremylt 264a2e7687Sjeremylt for (CeedInt i=0; i<impl->numein+impl->numeout; i++) { 274a2e7687Sjeremylt ierr = CeedElemRestrictionDestroy(&impl->blkrestr[i]); CeedChk(ierr); 284a2e7687Sjeremylt ierr = CeedVectorDestroy(&impl->evecs[i]); CeedChk(ierr); 294a2e7687Sjeremylt } 304a2e7687Sjeremylt ierr = CeedFree(&impl->blkrestr); CeedChk(ierr); 314a2e7687Sjeremylt ierr = CeedFree(&impl->evecs); CeedChk(ierr); 324a2e7687Sjeremylt ierr = CeedFree(&impl->edata); CeedChk(ierr); 334a2e7687Sjeremylt 344a2e7687Sjeremylt for (CeedInt i=0; i<impl->numqin+impl->numqout; i++) { 354a2e7687Sjeremylt ierr = CeedFree(&impl->qdata_alloc[i]); CeedChk(ierr); 364a2e7687Sjeremylt } 374a2e7687Sjeremylt ierr = CeedFree(&impl->qdata_alloc); CeedChk(ierr); 384a2e7687Sjeremylt ierr = CeedFree(&impl->qdata); CeedChk(ierr); 394a2e7687Sjeremylt 404a2e7687Sjeremylt ierr = CeedFree(&impl->indata); CeedChk(ierr); 414a2e7687Sjeremylt ierr = CeedFree(&impl->outdata); CeedChk(ierr); 424a2e7687Sjeremylt 43*fe2413ffSjeremylt ierr = CeedFree(&impl); CeedChk(ierr); 444a2e7687Sjeremylt return 0; 454a2e7687Sjeremylt } 464a2e7687Sjeremylt 474a2e7687Sjeremylt /* 484a2e7687Sjeremylt Setup infields or outfields 494a2e7687Sjeremylt */ 50*fe2413ffSjeremylt static int CeedOperatorSetupFields_Blocked(CeedQFunction qf, CeedOperator op, 51*fe2413ffSjeremylt bool inOrOut, 524a2e7687Sjeremylt CeedElemRestriction *blkrestr, 534a2e7687Sjeremylt CeedVector *evecs, CeedScalar **qdata, 544a2e7687Sjeremylt CeedScalar **qdata_alloc, CeedScalar **indata, 554a2e7687Sjeremylt CeedInt starti, CeedInt startq, 564a2e7687Sjeremylt CeedInt numfields, CeedInt Q) { 574a2e7687Sjeremylt CeedInt dim, ierr, iq=startq, ncomp; 58d1bcdac9Sjeremylt CeedBasis basis; 59d1bcdac9Sjeremylt CeedElemRestriction r; 60*fe2413ffSjeremylt CeedOperatorField *ofields; 61*fe2413ffSjeremylt CeedQFunctionField *qfields; 62*fe2413ffSjeremylt if (inOrOut) { 63*fe2413ffSjeremylt ierr = CeedOperatorGetFields(op, NULL, &ofields); 64*fe2413ffSjeremylt CeedChk(ierr); 65*fe2413ffSjeremylt ierr = CeedQFunctionGetFields(qf, NULL, &qfields); 66*fe2413ffSjeremylt CeedChk(ierr); 67*fe2413ffSjeremylt } else { 68*fe2413ffSjeremylt ierr = CeedOperatorGetFields(op, &ofields, NULL); 69*fe2413ffSjeremylt CeedChk(ierr); 70*fe2413ffSjeremylt ierr = CeedQFunctionGetFields(qf, &qfields, NULL); 71*fe2413ffSjeremylt CeedChk(ierr); 72*fe2413ffSjeremylt } 734a2e7687Sjeremylt const CeedInt blksize = 8; 744a2e7687Sjeremylt 754a2e7687Sjeremylt // Loop over fields 764a2e7687Sjeremylt for (CeedInt i=0; i<numfields; i++) { 77d1bcdac9Sjeremylt CeedEvalMode emode; 78d1bcdac9Sjeremylt ierr = CeedQFunctionFieldGetEvalMode(qfields[i], &emode); CeedChk(ierr); 794a2e7687Sjeremylt 804a2e7687Sjeremylt if (emode != CEED_EVAL_WEIGHT) { 81d1bcdac9Sjeremylt ierr = CeedOperatorFieldGetElemRestriction(ofields[i], &r); 82d1bcdac9Sjeremylt CeedChk(ierr); 83*fe2413ffSjeremylt CeedElemRestriction_Ref *data; 84*fe2413ffSjeremylt ierr = CeedElemRestrictionGetData(r, (void *)&data); 854ce2993fSjeremylt Ceed ceed; 864ce2993fSjeremylt ierr = CeedElemRestrictionGetCeed(r, &ceed); CeedChk(ierr); 874ce2993fSjeremylt CeedInt nelem, elemsize, ndof, ncomp; 884ce2993fSjeremylt ierr = CeedElemRestrictionGetNumElements(r, &nelem); CeedChk(ierr); 894ce2993fSjeremylt ierr = CeedElemRestrictionGetElementSize(r, &elemsize); CeedChk(ierr); 904ce2993fSjeremylt ierr = CeedElemRestrictionGetNumDoF(r, &ndof); CeedChk(ierr); 914ce2993fSjeremylt ierr = CeedElemRestrictionGetNumComponents(r, &ncomp); CeedChk(ierr); 924ce2993fSjeremylt ierr = CeedElemRestrictionCreateBlocked(ceed, nelem, elemsize, 934ce2993fSjeremylt blksize, ndof, ncomp, 944a2e7687Sjeremylt CEED_MEM_HOST, CEED_COPY_VALUES, 954a2e7687Sjeremylt data->indices, &blkrestr[i+starti]); 964a2e7687Sjeremylt CeedChk(ierr); 974a2e7687Sjeremylt ierr = CeedElemRestrictionCreateVector(blkrestr[i+starti], NULL, 984a2e7687Sjeremylt &evecs[i+starti]); 994a2e7687Sjeremylt CeedChk(ierr); 1004a2e7687Sjeremylt } 1014a2e7687Sjeremylt 1024a2e7687Sjeremylt switch(emode) { 1034a2e7687Sjeremylt case CEED_EVAL_NONE: 1044a2e7687Sjeremylt break; // No action 1054a2e7687Sjeremylt case CEED_EVAL_INTERP: 106d1bcdac9Sjeremylt ierr = CeedQFunctionFieldGetNumComponents(qfields[i], &ncomp); 107d1bcdac9Sjeremylt CeedChk(ierr); 1084a2e7687Sjeremylt ierr = CeedMalloc(Q*ncomp*blksize, &qdata_alloc[iq]); CeedChk(ierr); 1094a2e7687Sjeremylt qdata[i + starti] = qdata_alloc[iq]; 1104a2e7687Sjeremylt iq++; 1114a2e7687Sjeremylt break; 1124a2e7687Sjeremylt case CEED_EVAL_GRAD: 113d1bcdac9Sjeremylt ierr = CeedOperatorFieldGetBasis(ofields[i], &basis); CeedChk(ierr); 114d1bcdac9Sjeremylt ierr = CeedQFunctionFieldGetNumComponents(qfields[i], &ncomp); 115d1bcdac9Sjeremylt ierr = CeedBasisGetDimension(basis, &dim); CeedChk(ierr); 1164a2e7687Sjeremylt ierr = CeedMalloc(Q*ncomp*dim*blksize, &qdata_alloc[iq]); CeedChk(ierr); 1174a2e7687Sjeremylt qdata[i + starti] = qdata_alloc[iq]; 1184a2e7687Sjeremylt iq++; 1194a2e7687Sjeremylt break; 1204a2e7687Sjeremylt case CEED_EVAL_WEIGHT: // Only on input fields 121d1bcdac9Sjeremylt ierr = CeedOperatorFieldGetBasis(ofields[i], &basis); CeedChk(ierr); 1224a2e7687Sjeremylt ierr = CeedMalloc(Q*blksize, &qdata_alloc[iq]); CeedChk(ierr); 123d1bcdac9Sjeremylt ierr = CeedBasisApply(basis, blksize, CEED_NOTRANSPOSE, 1244a2e7687Sjeremylt CEED_EVAL_WEIGHT, NULL, qdata_alloc[iq]); CeedChk(ierr); 1254a2e7687Sjeremylt qdata[i] = qdata_alloc[iq]; 1264a2e7687Sjeremylt indata[i] = qdata[i]; 1274a2e7687Sjeremylt iq++; 1284a2e7687Sjeremylt break; 1294a2e7687Sjeremylt case CEED_EVAL_DIV: 1304a2e7687Sjeremylt break; // Not implimented 1314a2e7687Sjeremylt case CEED_EVAL_CURL: 1324a2e7687Sjeremylt break; // Not implimented 1334a2e7687Sjeremylt } 1344a2e7687Sjeremylt } 1354a2e7687Sjeremylt return 0; 1364a2e7687Sjeremylt } 1374a2e7687Sjeremylt 1384a2e7687Sjeremylt /* 1394a2e7687Sjeremylt CeedOperator needs to connect all the named fields (be they active or passive) 1404a2e7687Sjeremylt to the named inputs and outputs of its CeedQFunction. 1414a2e7687Sjeremylt */ 1424a2e7687Sjeremylt static int CeedOperatorSetup_Blocked(CeedOperator op) { 1434a2e7687Sjeremylt int ierr; 1444ce2993fSjeremylt bool setupdone; 1454ce2993fSjeremylt ierr = CeedOperatorGetSetupStatus(op, &setupdone); CeedChk(ierr); 1464ce2993fSjeremylt if (setupdone) return 0; 1474ce2993fSjeremylt CeedOperator_Blocked *impl; 1484ce2993fSjeremylt ierr = CeedOperatorGetData(op, (void*)&impl); CeedChk(ierr); 1494ce2993fSjeremylt CeedQFunction qf; 1504ce2993fSjeremylt ierr = CeedOperatorGetQFunction(op, &qf); CeedChk(ierr); 1514ce2993fSjeremylt CeedInt Q, numinputfields, numoutputfields; 1524ce2993fSjeremylt ierr = CeedOperatorGetNumQuadraturePoints(op, &Q); CeedChk(ierr); 1534a2e7687Sjeremylt ierr= CeedQFunctionGetNumArgs(qf, &numinputfields, &numoutputfields); 1544a2e7687Sjeremylt CeedChk(ierr); 155d1bcdac9Sjeremylt CeedOperatorField *opinputfields, *opoutputfields; 156d1bcdac9Sjeremylt ierr = CeedOperatorGetFields(op, &opinputfields, &opoutputfields); 157d1bcdac9Sjeremylt CeedChk(ierr); 158d1bcdac9Sjeremylt CeedQFunctionField *qfinputfields, *qfoutputfields; 159d1bcdac9Sjeremylt ierr = CeedQFunctionGetFields(qf, &qfinputfields, &qfoutputfields); 160d1bcdac9Sjeremylt CeedChk(ierr); 161d1bcdac9Sjeremylt CeedEvalMode emode; 1624ce2993fSjeremylt 1634ce2993fSjeremylt // Count infield and outfield array sizes and evectors 1644a2e7687Sjeremylt impl->numein = numinputfields; 1654a2e7687Sjeremylt for (CeedInt i=0; i<numinputfields; i++) { 166d1bcdac9Sjeremylt ierr = CeedQFunctionFieldGetEvalMode(qfinputfields[i], &emode); 167d1bcdac9Sjeremylt CeedChk(ierr); 1684a2e7687Sjeremylt impl->numqin += !!(emode & CEED_EVAL_INTERP) + !!(emode & CEED_EVAL_GRAD) + 1694a2e7687Sjeremylt !!(emode & CEED_EVAL_WEIGHT); 1704a2e7687Sjeremylt } 1714a2e7687Sjeremylt impl->numeout = numoutputfields; 1724a2e7687Sjeremylt for (CeedInt i=0; i<numoutputfields; i++) { 173d1bcdac9Sjeremylt ierr = CeedQFunctionFieldGetEvalMode(qfoutputfields[i], &emode); 174d1bcdac9Sjeremylt CeedChk(ierr); 1754a2e7687Sjeremylt impl->numqout += !!(emode & CEED_EVAL_INTERP) + !!(emode & CEED_EVAL_GRAD); 1764a2e7687Sjeremylt } 1774a2e7687Sjeremylt 1784a2e7687Sjeremylt // Allocate 1794a2e7687Sjeremylt ierr = CeedCalloc(impl->numein + impl->numeout, &impl->blkrestr); 1804a2e7687Sjeremylt CeedChk(ierr); 1814a2e7687Sjeremylt ierr = CeedCalloc(impl->numein + impl->numeout, &impl->evecs); 1824a2e7687Sjeremylt CeedChk(ierr); 1834a2e7687Sjeremylt ierr = CeedCalloc(impl->numein + impl->numeout, &impl->edata); 1844a2e7687Sjeremylt CeedChk(ierr); 1854a2e7687Sjeremylt 1864a2e7687Sjeremylt ierr = CeedCalloc(impl->numqin + impl->numqout, &impl->qdata_alloc); 1874a2e7687Sjeremylt CeedChk(ierr); 1884a2e7687Sjeremylt ierr = CeedCalloc(numinputfields + numoutputfields, &impl->qdata); 1894a2e7687Sjeremylt CeedChk(ierr); 1904a2e7687Sjeremylt 1914a2e7687Sjeremylt ierr = CeedCalloc(16, &impl->indata); CeedChk(ierr); 1924a2e7687Sjeremylt ierr = CeedCalloc(16, &impl->outdata); CeedChk(ierr); 1934a2e7687Sjeremylt // Set up infield and outfield pointer arrays 1944a2e7687Sjeremylt // Infields 195*fe2413ffSjeremylt ierr = CeedOperatorSetupFields_Blocked(qf, op, 0, 1964a2e7687Sjeremylt impl->blkrestr, impl->evecs, 1974a2e7687Sjeremylt impl->qdata, impl->qdata_alloc, 1984a2e7687Sjeremylt impl->indata, 0, 1994a2e7687Sjeremylt 0, numinputfields, Q); 2004a2e7687Sjeremylt CeedChk(ierr); 2014a2e7687Sjeremylt // Outfields 202*fe2413ffSjeremylt ierr = CeedOperatorSetupFields_Blocked(qf, op, 1, 2034a2e7687Sjeremylt impl->blkrestr, impl->evecs, 2044a2e7687Sjeremylt impl->qdata, impl->qdata_alloc, 2054a2e7687Sjeremylt impl->indata, numinputfields, 2064a2e7687Sjeremylt impl->numqin, numoutputfields, Q); 2074a2e7687Sjeremylt CeedChk(ierr); 2084a2e7687Sjeremylt // Input Qvecs 2094a2e7687Sjeremylt for (CeedInt i=0; i<numinputfields; i++) { 210d1bcdac9Sjeremylt ierr = CeedQFunctionFieldGetEvalMode(qfinputfields[i], &emode); 211d1bcdac9Sjeremylt CeedChk(ierr); 2124a2e7687Sjeremylt if ((emode != CEED_EVAL_NONE) && (emode != CEED_EVAL_WEIGHT)) 2134a2e7687Sjeremylt impl->indata[i] = impl->qdata[i]; 2144a2e7687Sjeremylt } 2154a2e7687Sjeremylt // Output Qvecs 2164a2e7687Sjeremylt for (CeedInt i=0; i<numoutputfields; i++) { 217d1bcdac9Sjeremylt ierr = CeedQFunctionFieldGetEvalMode(qfoutputfields[i], &emode); 218d1bcdac9Sjeremylt CeedChk(ierr); 2194a2e7687Sjeremylt if (emode != CEED_EVAL_NONE) 2204a2e7687Sjeremylt impl->outdata[i] = impl->qdata[i + numinputfields]; 2214a2e7687Sjeremylt } 2224a2e7687Sjeremylt 2234ce2993fSjeremylt ierr = CeedOperatorSetSetupDone(op); CeedChk(ierr); 2244a2e7687Sjeremylt 2254a2e7687Sjeremylt return 0; 2264a2e7687Sjeremylt } 2274a2e7687Sjeremylt 2284a2e7687Sjeremylt static int CeedOperatorApply_Blocked(CeedOperator op, CeedVector invec, 2294a2e7687Sjeremylt CeedVector outvec, CeedRequest *request) { 2304a2e7687Sjeremylt int ierr; 2314ce2993fSjeremylt CeedOperator_Blocked *impl; 2324ce2993fSjeremylt ierr = CeedOperatorGetData(op, (void*)&impl); CeedChk(ierr); 2334ce2993fSjeremylt const CeedInt blksize = 8; 234d1bcdac9Sjeremylt CeedInt Q, elemsize, numinputfields, numoutputfields, numelements, ncomp; 2354ce2993fSjeremylt ierr = CeedOperatorGetNumElements(op, &numelements); CeedChk(ierr); 2364ce2993fSjeremylt ierr = CeedOperatorGetNumQuadraturePoints(op, &Q); CeedChk(ierr); 2374ce2993fSjeremylt CeedInt nblks = (numelements/blksize) + !!(numelements%blksize); 2384ce2993fSjeremylt CeedQFunction qf; 2394ce2993fSjeremylt ierr = CeedOperatorGetQFunction(op, &qf); CeedChk(ierr); 2404ce2993fSjeremylt ierr= CeedQFunctionGetNumArgs(qf, &numinputfields, &numoutputfields); 2414ce2993fSjeremylt CeedChk(ierr); 2424dccadb6Sjeremylt CeedTransposeMode lmode; 243d1bcdac9Sjeremylt CeedOperatorField *opinputfields, *opoutputfields; 244d1bcdac9Sjeremylt ierr = CeedOperatorGetFields(op, &opinputfields, &opoutputfields); 245d1bcdac9Sjeremylt CeedChk(ierr); 246d1bcdac9Sjeremylt CeedQFunctionField *qfinputfields, *qfoutputfields; 247d1bcdac9Sjeremylt ierr = CeedQFunctionGetFields(qf, &qfinputfields, &qfoutputfields); 248d1bcdac9Sjeremylt CeedChk(ierr); 249d1bcdac9Sjeremylt CeedEvalMode emode; 250d1bcdac9Sjeremylt CeedVector vec; 251d1bcdac9Sjeremylt CeedBasis basis; 252d1bcdac9Sjeremylt CeedElemRestriction Erestrict; 2534a2e7687Sjeremylt 2544a2e7687Sjeremylt // Setup 2554a2e7687Sjeremylt ierr = CeedOperatorSetup_Blocked(op); CeedChk(ierr); 2564a2e7687Sjeremylt 2574a2e7687Sjeremylt // Input Evecs and Restriction 2584a2e7687Sjeremylt for (CeedInt i=0; i<numinputfields; i++) { 259d1bcdac9Sjeremylt ierr = CeedQFunctionFieldGetEvalMode(qfinputfields[i], &emode); 260d1bcdac9Sjeremylt CeedChk(ierr); 2614a2e7687Sjeremylt if (emode == CEED_EVAL_WEIGHT) { // Skip 2624a2e7687Sjeremylt } else { 263d1bcdac9Sjeremylt // Get input vector 264d1bcdac9Sjeremylt ierr = CeedOperatorFieldGetVector(opinputfields[i], &vec); CeedChk(ierr); 265d1bcdac9Sjeremylt if (vec == CEED_VECTOR_ACTIVE) 266d1bcdac9Sjeremylt vec = invec; 2674a2e7687Sjeremylt // Restrict 2684dccadb6Sjeremylt ierr = CeedOperatorFieldGetLMode(opinputfields[i], &lmode); CeedChk(ierr); 2694a2e7687Sjeremylt ierr = CeedElemRestrictionApply(impl->blkrestr[i], CEED_NOTRANSPOSE, 270d1bcdac9Sjeremylt lmode, vec, impl->evecs[i], 2714a2e7687Sjeremylt request); CeedChk(ierr); CeedChk(ierr); 2724a2e7687Sjeremylt // Get evec 2734a2e7687Sjeremylt ierr = CeedVectorGetArrayRead(impl->evecs[i], CEED_MEM_HOST, 2744a2e7687Sjeremylt (const CeedScalar **) &impl->edata[i]); 2754a2e7687Sjeremylt CeedChk(ierr); 2764a2e7687Sjeremylt } 2774a2e7687Sjeremylt } 2784a2e7687Sjeremylt 2794a2e7687Sjeremylt // Output Evecs 2804a2e7687Sjeremylt for (CeedInt i=0; i<numoutputfields; i++) { 2814a2e7687Sjeremylt ierr = CeedVectorGetArray(impl->evecs[i+impl->numein], CEED_MEM_HOST, 2824a2e7687Sjeremylt &impl->edata[i + numinputfields]); CeedChk(ierr); 2834a2e7687Sjeremylt } 2844a2e7687Sjeremylt 2854a2e7687Sjeremylt // Loop through elements 2864a2e7687Sjeremylt for (CeedInt e=0; e<nblks*blksize; e+=blksize) { 2874a2e7687Sjeremylt // Input basis apply if needed 2884a2e7687Sjeremylt for (CeedInt i=0; i<numinputfields; i++) { 2894a2e7687Sjeremylt // Get elemsize, emode, ncomp 290d1bcdac9Sjeremylt ierr = CeedOperatorFieldGetElemRestriction(opinputfields[i], &Erestrict); 291d1bcdac9Sjeremylt CeedChk(ierr); 292d1bcdac9Sjeremylt ierr = CeedElemRestrictionGetElementSize(Erestrict, &elemsize); 293d1bcdac9Sjeremylt CeedChk(ierr); 294d1bcdac9Sjeremylt ierr = CeedQFunctionFieldGetEvalMode(qfinputfields[i], &emode); 295d1bcdac9Sjeremylt CeedChk(ierr); 296d1bcdac9Sjeremylt ierr = CeedQFunctionFieldGetNumComponents(qfinputfields[i], &ncomp); 297d1bcdac9Sjeremylt CeedChk(ierr); 2984a2e7687Sjeremylt // Basis action 2994a2e7687Sjeremylt switch(emode) { 3004a2e7687Sjeremylt case CEED_EVAL_NONE: 3014a2e7687Sjeremylt impl->indata[i] = &impl->edata[i][e*Q*ncomp]; 3024a2e7687Sjeremylt break; 3034a2e7687Sjeremylt case CEED_EVAL_INTERP: 304d1bcdac9Sjeremylt ierr = CeedOperatorFieldGetBasis(opinputfields[i], &basis); 3054a2e7687Sjeremylt CeedChk(ierr); 306d1bcdac9Sjeremylt ierr = CeedBasisApply(basis, blksize, CEED_NOTRANSPOSE, 307d1bcdac9Sjeremylt CEED_EVAL_INTERP, &impl->edata[i][e*elemsize*ncomp], 308d1bcdac9Sjeremylt impl->qdata[i]); CeedChk(ierr); 3094a2e7687Sjeremylt break; 3104a2e7687Sjeremylt case CEED_EVAL_GRAD: 311d1bcdac9Sjeremylt ierr = CeedOperatorFieldGetBasis(opinputfields[i], &basis); 3124a2e7687Sjeremylt CeedChk(ierr); 313d1bcdac9Sjeremylt ierr = CeedBasisApply(basis, blksize, CEED_NOTRANSPOSE, 314d1bcdac9Sjeremylt CEED_EVAL_GRAD, &impl->edata[i][e*elemsize*ncomp], 315d1bcdac9Sjeremylt impl->qdata[i]); CeedChk(ierr); 3164a2e7687Sjeremylt break; 3174a2e7687Sjeremylt case CEED_EVAL_WEIGHT: 3184a2e7687Sjeremylt break; // No action 3194a2e7687Sjeremylt case CEED_EVAL_DIV: 3204a2e7687Sjeremylt break; // Not implimented 3214a2e7687Sjeremylt case CEED_EVAL_CURL: 3224a2e7687Sjeremylt break; // Not implimented 3234a2e7687Sjeremylt } 3244a2e7687Sjeremylt } 3254a2e7687Sjeremylt 3264a2e7687Sjeremylt // Output pointers 3274a2e7687Sjeremylt for (CeedInt i=0; i<numoutputfields; i++) { 328d1bcdac9Sjeremylt ierr = CeedQFunctionFieldGetEvalMode(qfoutputfields[i], &emode); 329d1bcdac9Sjeremylt CeedChk(ierr); 3304a2e7687Sjeremylt if (emode == CEED_EVAL_NONE) { 331d1bcdac9Sjeremylt ierr = CeedQFunctionFieldGetNumComponents(qfoutputfields[i], &ncomp); 332d1bcdac9Sjeremylt CeedChk(ierr); 3334a2e7687Sjeremylt impl->outdata[i] = &impl->edata[i + numinputfields][e*Q*ncomp]; 3344a2e7687Sjeremylt } 3354a2e7687Sjeremylt } 3364a2e7687Sjeremylt // Q function 3374ce2993fSjeremylt ierr = CeedQFunctionApply(qf, Q*blksize, 3384a2e7687Sjeremylt (const CeedScalar * const*) impl->indata, 3394a2e7687Sjeremylt impl->outdata); CeedChk(ierr); 3404a2e7687Sjeremylt 3414a2e7687Sjeremylt // Output basis apply if needed 3424a2e7687Sjeremylt for (CeedInt i=0; i<numoutputfields; i++) { 3434a2e7687Sjeremylt // Get elemsize, emode, ncomp 344d1bcdac9Sjeremylt ierr = CeedOperatorFieldGetElemRestriction(opoutputfields[i], &Erestrict); 345d1bcdac9Sjeremylt CeedChk(ierr); 346d1bcdac9Sjeremylt ierr = CeedElemRestrictionGetElementSize(Erestrict, &elemsize); 347d1bcdac9Sjeremylt CeedChk(ierr); 348d1bcdac9Sjeremylt ierr = CeedQFunctionFieldGetEvalMode(qfoutputfields[i], &emode); 349d1bcdac9Sjeremylt CeedChk(ierr); 350d1bcdac9Sjeremylt ierr = CeedQFunctionFieldGetNumComponents(qfoutputfields[i], &ncomp); 351d1bcdac9Sjeremylt CeedChk(ierr); 3524a2e7687Sjeremylt // Basis action 3534a2e7687Sjeremylt switch(emode) { 3544a2e7687Sjeremylt case CEED_EVAL_NONE: 3554a2e7687Sjeremylt break; // No action 3564a2e7687Sjeremylt case CEED_EVAL_INTERP: 357d1bcdac9Sjeremylt ierr = CeedOperatorFieldGetBasis(opoutputfields[i], &basis); 358d1bcdac9Sjeremylt CeedChk(ierr); 359d1bcdac9Sjeremylt ierr = CeedBasisApply(basis, blksize, CEED_TRANSPOSE, 3604a2e7687Sjeremylt CEED_EVAL_INTERP, impl->outdata[i], 3614a2e7687Sjeremylt &impl->edata[i + numinputfields][e*elemsize*ncomp]); 3624a2e7687Sjeremylt CeedChk(ierr); 3634a2e7687Sjeremylt break; 3644a2e7687Sjeremylt case CEED_EVAL_GRAD: 365d1bcdac9Sjeremylt ierr = CeedOperatorFieldGetBasis(opoutputfields[i], &basis); 366d1bcdac9Sjeremylt CeedChk(ierr); 367d1bcdac9Sjeremylt ierr = CeedBasisApply(basis, blksize, CEED_TRANSPOSE, 3684a2e7687Sjeremylt CEED_EVAL_GRAD, 3694a2e7687Sjeremylt impl->outdata[i], &impl->edata[i + numinputfields][e*elemsize*ncomp]); 3704a2e7687Sjeremylt CeedChk(ierr); 3714a2e7687Sjeremylt break; 3724ce2993fSjeremylt case CEED_EVAL_WEIGHT: { 3734ce2993fSjeremylt Ceed ceed; 3744ce2993fSjeremylt ierr = CeedOperatorGetCeed(op, &ceed); CeedChk(ierr); 3754ce2993fSjeremylt return CeedError(ceed, 1, 3764a2e7687Sjeremylt "CEED_EVAL_WEIGHT cannot be an output evaluation mode"); 3774a2e7687Sjeremylt break; // Should not occur 3784ce2993fSjeremylt } 3794a2e7687Sjeremylt case CEED_EVAL_DIV: 3804a2e7687Sjeremylt break; // Not implimented 3814a2e7687Sjeremylt case CEED_EVAL_CURL: 3824a2e7687Sjeremylt break; // Not implimented 3834a2e7687Sjeremylt } 3844a2e7687Sjeremylt } 3854a2e7687Sjeremylt } 3864a2e7687Sjeremylt 3874a2e7687Sjeremylt // Zero lvecs 388d1bcdac9Sjeremylt for (CeedInt i=0; i<numoutputfields; i++) { 389d1bcdac9Sjeremylt ierr = CeedOperatorFieldGetVector(opoutputfields[i], &vec); CeedChk(ierr); 390d1bcdac9Sjeremylt if (vec == CEED_VECTOR_ACTIVE) 391d1bcdac9Sjeremylt vec = outvec; 392d1bcdac9Sjeremylt ierr = CeedVectorSetValue(vec, 0.0); CeedChk(ierr); 3934a2e7687Sjeremylt } 3944a2e7687Sjeremylt 3954a2e7687Sjeremylt // Output restriction 3964a2e7687Sjeremylt for (CeedInt i=0; i<numoutputfields; i++) { 3974a2e7687Sjeremylt // Restore evec 3984a2e7687Sjeremylt ierr = CeedVectorRestoreArray(impl->evecs[i+impl->numein], 3994a2e7687Sjeremylt &impl->edata[i + numinputfields]); CeedChk(ierr); 400d1bcdac9Sjeremylt // Get output vector 401d1bcdac9Sjeremylt ierr = CeedOperatorFieldGetVector(opoutputfields[i], &vec); CeedChk(ierr); 4024a2e7687Sjeremylt // Active 403d1bcdac9Sjeremylt if (vec == CEED_VECTOR_ACTIVE) 404d1bcdac9Sjeremylt vec = outvec; 4054a2e7687Sjeremylt // Restrict 4064dccadb6Sjeremylt ierr = CeedOperatorFieldGetLMode(opoutputfields[i], &lmode); CeedChk(ierr); 4074a2e7687Sjeremylt ierr = CeedElemRestrictionApply(impl->blkrestr[i+impl->numein], CEED_TRANSPOSE, 408d1bcdac9Sjeremylt lmode, impl->evecs[i+impl->numein], vec, 4094a2e7687Sjeremylt request); CeedChk(ierr); 410d1bcdac9Sjeremylt 4114a2e7687Sjeremylt } 4124a2e7687Sjeremylt 4134a2e7687Sjeremylt // Restore input arrays 4144a2e7687Sjeremylt for (CeedInt i=0; i<numinputfields; i++) { 415d1bcdac9Sjeremylt ierr = CeedQFunctionFieldGetEvalMode(qfinputfields[i], &emode); 416d1bcdac9Sjeremylt CeedChk(ierr); 4174a2e7687Sjeremylt if (emode == CEED_EVAL_WEIGHT) { // Skip 4184a2e7687Sjeremylt } else { 4194a2e7687Sjeremylt ierr = CeedVectorRestoreArrayRead(impl->evecs[i], 420d1bcdac9Sjeremylt (const CeedScalar **) &impl->edata[i]); 421d1bcdac9Sjeremylt CeedChk(ierr); 4224a2e7687Sjeremylt } 4234a2e7687Sjeremylt } 4244a2e7687Sjeremylt 4254a2e7687Sjeremylt return 0; 4264a2e7687Sjeremylt } 4274a2e7687Sjeremylt 4284a2e7687Sjeremylt int CeedOperatorCreate_Blocked(CeedOperator op) { 4294a2e7687Sjeremylt int ierr; 430*fe2413ffSjeremylt Ceed ceed; 431*fe2413ffSjeremylt ierr = CeedOperatorGetCeed(op, &ceed); CeedChk(ierr); 4324ce2993fSjeremylt CeedOperator_Blocked *impl; 4334a2e7687Sjeremylt 4344a2e7687Sjeremylt ierr = CeedCalloc(1, &impl); CeedChk(ierr); 435*fe2413ffSjeremylt ierr = CeedOperatorSetData(op, (void *)&impl); 436*fe2413ffSjeremylt 437*fe2413ffSjeremylt ierr = CeedSetBackendFunction(ceed, "Operator", op, "Apply", 438*fe2413ffSjeremylt CeedOperatorApply_Blocked); CeedChk(ierr); 439*fe2413ffSjeremylt ierr = CeedSetBackendFunction(ceed, "Operator", op, "Destroy", 440*fe2413ffSjeremylt CeedOperatorDestroy_Blocked); CeedChk(ierr); 4414a2e7687Sjeremylt return 0; 4424a2e7687Sjeremylt } 443