Files
project_6_89d52222/upstream_ref/xllm/xllm/server/xllm_server.h
EX Engine 002f9879b2 ref(upstream): FULL TREE — Deep-Spark xllm (1470) + ds_vllm csrc/models (703)
Replaces cherry-picked upstream_ref with complete source trees.

xllm/ — Iluvatar official C++ inference engine (15MB, 1470 files)
  Complete: kernels → layers → models → runtime → scheduler → api
  Excluded: .git, binary images, third_party submodule checkouts

ds_vllm/ — Iluvatar official vllm fork (8MB, 703 files)
  Included: csrc/ (ALL CUDA kernels), fused_moe/, qwen3_5 model, _custom_ops
  Excluded: tests, benchmarks, docs, examples (not needed for reference)

Critical call chains now fully traceable:
  MoE: moe_topk_softmax_kernels.cuh → ixformer.h → fused_moe.cpp → layer
  GDN: qwen3_gated_delta_net_base.cpp → qwen3_5_gated_delta_net.cpp
  Attention: ixformer.h → xllm_paged_attention → attention.cpp
2026-08-10 02:54:03 +00:00

77 lines
2.5 KiB
C++

/* Copyright 2025 The xLLM Authors. All Rights Reserved.
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
https://github.com/jd-opensource/xllm/blob/main/LICENSE
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
==============================================================================*/
#pragma once
#include "api_service/api_service.h"
#include "core/distributed_runtime/collective_service.h"
#include "core/distributed_runtime/disagg_pd_service.h"
#include "core/distributed_runtime/pd_ooc_service.h"
#include "core/distributed_runtime/worker_service.h"
#include "core/framework/xtensor/xtensor_dist_service.h"
namespace xllm {
class XllmServer final {
public:
XllmServer();
~XllmServer();
bool start(std::unique_ptr<APIService> api_service);
bool start(std::unique_ptr<DisaggPDService> disagg_pd_service);
bool start(std::unique_ptr<PDOOCService> pd_ooc_service);
bool start(std::shared_ptr<CollectiveService> service,
const std::string& addr,
const std::string& server_name);
bool start(std::shared_ptr<WorkerService> service, const std::string& addr);
bool start(std::shared_ptr<XTensorDistService> service,
const std::string& addr);
void run();
void stop();
bool has_initialized() const { return has_initialized_; }
std::string listen_address() const { return listen_address_; }
int listen_port() const { return listen_port_; }
private:
DISALLOW_COPY_AND_ASSIGN(XllmServer);
bool create_server(google::protobuf::Service* service,
const std::string& addr,
int port,
const std::string& server_name);
private:
bool has_initialized_ = false;
int listen_port_ = -1;
std::string listen_address_;
std::unique_ptr<brpc::Server> server_;
std::unique_ptr<std::thread> running_thread_;
};
class XllmServerFactory {
public:
static std::unique_ptr<XllmServer> create_xllm_server() {
return std::make_unique<XllmServer>();
}
private:
XllmServerFactory() = default;
~XllmServerFactory() = default;
DISALLOW_COPY_AND_ASSIGN(XllmServerFactory);
};
} // namespace xllm