Replaces cherry-picked upstream_ref with complete source trees. xllm/ — Iluvatar official C++ inference engine (15MB, 1470 files) Complete: kernels → layers → models → runtime → scheduler → api Excluded: .git, binary images, third_party submodule checkouts ds_vllm/ — Iluvatar official vllm fork (8MB, 703 files) Included: csrc/ (ALL CUDA kernels), fused_moe/, qwen3_5 model, _custom_ops Excluded: tests, benchmarks, docs, examples (not needed for reference) Critical call chains now fully traceable: MoE: moe_topk_softmax_kernels.cuh → ixformer.h → fused_moe.cpp → layer GDN: qwen3_gated_delta_net_base.cpp → qwen3_5_gated_delta_net.cpp Attention: ixformer.h → xllm_paged_attention → attention.cpp
65 lines
2.2 KiB
C++
65 lines
2.2 KiB
C++
/* Copyright 2025 The xLLM Authors. All Rights Reserved.
|
|
|
|
Licensed under the Apache License, Version 2.0 (the "License");
|
|
you may not use this file except in compliance with the License.
|
|
You may obtain a copy of the License at
|
|
|
|
https://github.com/jd-opensource/xllm/blob/main/LICENSE
|
|
|
|
Unless required by applicable law or agreed to in writing, software
|
|
distributed under the License is distributed on an "AS IS" BASIS,
|
|
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
See the License for the specific language governing permissions and
|
|
limitations under the License.
|
|
==============================================================================*/
|
|
|
|
#include "call.h"
|
|
|
|
#include "core/common/constants.h"
|
|
|
|
namespace xllm {
|
|
|
|
Call::Call(brpc::Controller* controller) : controller_(controller) { init(); }
|
|
|
|
void Call::init() {
|
|
if (controller_->http_request().GetHeader("x-request-id")) {
|
|
x_request_id_ = *controller_->http_request().GetHeader("x-request-id");
|
|
} else if (controller_->http_request().GetHeader("x-ms-client-request-id")) {
|
|
x_request_id_ =
|
|
*controller_->http_request().GetHeader("x-ms-client-request-id");
|
|
}
|
|
|
|
if (controller_->http_request().GetHeader("x-request-time")) {
|
|
x_request_time_ = *controller_->http_request().GetHeader("x-request-time");
|
|
} else if (controller_->http_request().GetHeader("x-request-timems")) {
|
|
x_request_time_ =
|
|
*controller_->http_request().GetHeader("x-request-timems");
|
|
}
|
|
|
|
init_request_payload();
|
|
}
|
|
|
|
void Call::init_request_payload() {
|
|
const auto infer_content_len =
|
|
controller_->http_request().GetHeader(kInferContentLength);
|
|
const auto content_len =
|
|
controller_->http_request().GetHeader(kContentLength);
|
|
|
|
if (infer_content_len == nullptr || content_len == nullptr) return;
|
|
|
|
auto infer_len = std::stoul(*infer_content_len);
|
|
auto len = std::stoul(*content_len);
|
|
|
|
if (infer_len > len) {
|
|
LOG(ERROR) << " content length is invalid:"
|
|
<< " infer content len is " << infer_len
|
|
<< " , content length is " << len;
|
|
return;
|
|
}
|
|
|
|
controller_->request_attachment().copy_to(
|
|
&request_payload_, len - infer_len, infer_len);
|
|
}
|
|
|
|
} // namespace xllm
|