/** * This program is free software, you can redistribute it and/or modify it. * Copyright (c) 2025 Huawei Technologies Co., Ltd. * This file is a part of the CANN Open Software. * Licensed under CANN Open Software License Agreement Version 2.0 (the "License"). * Please refer to the License for details. You may not use this file except in compliance with the License. * THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, INCLUDING * BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. * See LICENSE in the root of the software repository for the full text of the License. */ #ifndef CAUSAL_CONV1D_FN_H #define CAUSAL_CONV1D_FN_H #include "causal_conv1d.h" namespace NsCausalConv1d { template class CausalConv1dFn : public CausalConv1d { public: __aicore__ inline void Init(GM_ADDR x, GM_ADDR weight, GM_ADDR bias, GM_ADDR convStates, GM_ADDR queryStartLoc, GM_ADDR cacheIndices, GM_ADDR initialStateMode, GM_ADDR numAcceptedTokens, GM_ADDR y, GM_ADDR workspace, const CausalConv1dTilingData *tilingData) { (void)numAcceptedTokens; this->ResetRuntimeState(tilingData); this->xGm.SetGlobalBuffer(reinterpret_cast<__gm__ T *>(x)); this->weightGm.SetGlobalBuffer(reinterpret_cast<__gm__ T *>(weight)); this->biasGm.SetGlobalBuffer(reinterpret_cast<__gm__ T *>(bias)); this->convStatesGm.SetGlobalBuffer(reinterpret_cast<__gm__ T *>(convStates)); if (tilingData->hasQueryStartLoc != 0) { if (tilingData->queryStartLocUseInt64 != 0) { this->queryStartLocGmInt64.SetGlobalBuffer(reinterpret_cast<__gm__ int64_t *>(queryStartLoc)); } else { this->queryStartLocGmInt32.SetGlobalBuffer(reinterpret_cast<__gm__ int32_t *>(queryStartLoc)); } } if (tilingData->hasCacheIndices != 0) { if (tilingData->cacheIndicesUseInt64 != 0) { this->cacheIndicesGmInt64.SetGlobalBuffer(reinterpret_cast<__gm__ int64_t *>(cacheIndices)); } else { this->cacheIndicesGmInt32.SetGlobalBuffer(reinterpret_cast<__gm__ int32_t *>(cacheIndices)); } } if (tilingData->hasInitialStateMode != 0) { if (tilingData->initialStateModeDtype == 2) { this->initialStateModeGmInt64.SetGlobalBuffer(reinterpret_cast<__gm__ int64_t *>(initialStateMode)); } else if (tilingData->initialStateModeDtype == 1) { this->initialStateModeGmInt32.SetGlobalBuffer(reinterpret_cast<__gm__ int32_t *>(initialStateMode)); } else { this->initialStateModeGmBool.SetGlobalBuffer(reinterpret_cast<__gm__ bool *>(initialStateMode)); } } this->yGm.SetGlobalBuffer(reinterpret_cast<__gm__ T *>(y)); if (tilingData->hasInitStateWorkspace != 0) { const uint64_t syncElems = static_cast(GetBlockNum()) * INIT_STATE_SYNCALL_NEED_SIZE; const uint64_t syncBytes = syncElems * sizeof(int32_t); const uint64_t workspaceElems = static_cast(tilingData->batch) * static_cast(tilingData->width - 1) * static_cast(tilingData->dim); this->initStateSyncGm_.SetGlobalBuffer(reinterpret_cast<__gm__ int32_t *>(workspace), syncElems); auto *workspaceBytes = reinterpret_cast<__gm__ uint8_t *>(workspace); this->initStateWorkspaceGm_.SetGlobalBuffer(reinterpret_cast<__gm__ T *>(workspaceBytes + syncBytes), workspaceElems); } this->InitSharedBuffersAndEvents(); } __aicore__ inline void Process() { this->ProcessVarlenTokenTiled(); this->ReleaseEvents(); } }; template __aicore__ inline void RunCausalConv1dFn(GM_ADDR x, GM_ADDR weight, GM_ADDR bias, GM_ADDR convStates, GM_ADDR queryStartLoc, GM_ADDR cacheIndices, GM_ADDR initialStateMode, GM_ADDR numAcceptedTokens, GM_ADDR y, GM_ADDR workspace, const CausalConv1dTilingData *tilingData) { CausalConv1dFn op; op.Init(x, weight, bias, convStates, queryStartLoc, cacheIndices, initialStateMode, numAcceptedTokens, y, workspace, tilingData); op.Process(); } } #endif