/** * This program is free software, you can redistribute it and/or modify it. * Copyright (c) 2025 Huawei Technologies Co., Ltd. * This file is a part of the CANN Open Software. * Licensed under CANN Open Software License Agreement Version 2.0 (the "License"). * Please refer to the License for details. You may not use this file except in compliance with the License. * THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, INCLUDING * BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. * See LICENSE in the root of the software repository for the full text of the License. */ #ifndef CAUSAL_CONV1D_UPDATE_H #define CAUSAL_CONV1D_UPDATE_H #include "causal_conv1d.h" namespace NsCausalConv1d { template class CausalConv1dUpdate : public CausalConv1d { public: __aicore__ inline void Init(GM_ADDR x, GM_ADDR weight, GM_ADDR bias, GM_ADDR convStates, GM_ADDR queryStartLoc, GM_ADDR cacheIndices, GM_ADDR, GM_ADDR numAcceptedTokens, GM_ADDR y, GM_ADDR workspace, const CausalConv1dTilingData *tilingData) { (void)workspace; this->ResetRuntimeState(tilingData); this->xGm.SetGlobalBuffer(reinterpret_cast<__gm__ T *>(x)); this->weightGm.SetGlobalBuffer(reinterpret_cast<__gm__ T *>(weight)); this->biasGm.SetGlobalBuffer(reinterpret_cast<__gm__ T *>(bias)); this->convStatesGm.SetGlobalBuffer(reinterpret_cast<__gm__ T *>(convStates)); if (tilingData->hasQueryStartLoc != 0) { if (tilingData->queryStartLocUseInt64 != 0) { this->queryStartLocGmInt64.SetGlobalBuffer(reinterpret_cast<__gm__ int64_t *>(queryStartLoc)); } else { this->queryStartLocGmInt32.SetGlobalBuffer(reinterpret_cast<__gm__ int32_t *>(queryStartLoc)); } } if (tilingData->hasCacheIndices != 0) { if (tilingData->cacheIndicesUseInt64 != 0) { this->cacheIndicesGmInt64.SetGlobalBuffer(reinterpret_cast<__gm__ int64_t *>(cacheIndices)); } else { this->cacheIndicesGmInt32.SetGlobalBuffer(reinterpret_cast<__gm__ int32_t *>(cacheIndices)); } } if (tilingData->hasNumAcceptedTokens != 0) { if (tilingData->numAcceptedTokensUseInt64 != 0) { this->numAcceptedTokensGmInt64.SetGlobalBuffer(reinterpret_cast<__gm__ int64_t *>(numAcceptedTokens)); } else { this->numAcceptedTokensGmInt32.SetGlobalBuffer(reinterpret_cast<__gm__ int32_t *>(numAcceptedTokens)); } } this->yGm.SetGlobalBuffer(reinterpret_cast<__gm__ T *>(y)); this->InitSharedBuffersAndEvents(); } __aicore__ inline void Process() { const CausalConv1dTilingData *tilingData = this->GetTilingData(); const int32_t dim = tilingData->dim; const int32_t baseDimCnt = static_cast(tilingData->baseDimCnt); const int32_t width = static_cast(tilingData->width); const int32_t baseDim = static_cast(tilingData->baseDim); if (baseDim <= 0 || baseDimCnt <= 0 || baseDim > MAX_BLOCK_DIM || width < 2 || width > MAX_WIDTH || dim <= 0 || tilingData->batch <= 0) { this->ReleaseEvents(); return; } this->ProcessDefault(); this->ReleaseEvents(); } }; template __aicore__ inline void RunCausalConv1dUpdate(GM_ADDR x, GM_ADDR weight, GM_ADDR bias, GM_ADDR convStates, GM_ADDR queryStartLoc, GM_ADDR cacheIndices, GM_ADDR initialStateMode, GM_ADDR numAcceptedTokens, GM_ADDR y, GM_ADDR workspace, const CausalConv1dTilingData *tilingData) { CausalConv1dUpdate op; op.Init(x, weight, bias, convStates, queryStartLoc, cacheIndices, initialStateMode, numAcceptedTokens, y, workspace, tilingData); op.Process(); } } // namespace NsCausalConv1d #endif // CAUSAL_CONV1D_UPDATE_H