Compare commits
10 Commits
810874ddb8
...
2c6cabcc63
| Author | SHA1 | Date | |
|---|---|---|---|
| 2c6cabcc63 | |||
| 0597fa6c6d | |||
| a5abd66e21 | |||
| 82c1463b0c | |||
| 465a9c8995 | |||
| b41e7cd899 | |||
| 61e169fe18 | |||
| c8cc9401a4 | |||
| 1902c81fdd | |||
| f89bc60d59 |
@@ -42,7 +42,8 @@ cp ./paged_attn.py /usr/local/corex/lib/python3/dist-packages/vllm/attention/ops
|
|||||||
python3 ./patch_model_runner.py
|
python3 ./patch_model_runner.py
|
||||||
|
|
||||||
# --- transformers: Qwen3_5 tokenizer / model files --------------------------
|
# --- transformers: Qwen3_5 tokenizer / model files --------------------------
|
||||||
pip install transformers==4.55.3 -i https://pypi.tuna.tsinghua.edu.cn/simple
|
pip install transformers==4.55.3 -i https://pypi.tuna.tsinghua.edu.cn/simple || \
|
||||||
|
echo "[patch_ops] WARN: transformers==4.55.3 install failed; continuing with installed transformers."
|
||||||
cp -r ./qwen3_5 /usr/local/lib/python3.10/site-packages/transformers/models/
|
cp -r ./qwen3_5 /usr/local/lib/python3.10/site-packages/transformers/models/
|
||||||
cp -r ./qwen3_5_moe /usr/local/lib/python3.10/site-packages/transformers/models/
|
cp -r ./qwen3_5_moe /usr/local/lib/python3.10/site-packages/transformers/models/
|
||||||
python3 ./patch_transformers_qwen3_5.py
|
python3 ./patch_transformers_qwen3_5.py
|
||||||
|
|||||||
@@ -29,6 +29,10 @@ flash attention kernel(ixformer / cudnnFlashAttnForward)。
|
|||||||
关闭选项),原意是防止 profiling OOM。但 _run_sdpa_fallback 已通过 Q-tiling
|
关闭选项),原意是防止 profiling OOM。但 _run_sdpa_fallback 已通过 Q-tiling
|
||||||
解决了该问题,chunked prefill 反而会把推理路径从 _run_sdpa_fallback 切换到
|
解决了该问题,chunked prefill 反而会把推理路径从 _run_sdpa_fallback 切换到
|
||||||
_forward_prefix_pytorch,属于不必要的行为变更,因此一并禁用该自动逻辑。
|
_forward_prefix_pytorch,属于不必要的行为变更,因此一并禁用该自动逻辑。
|
||||||
|
同时解除 EngineArgs.create_*_config 中强制 enforce_eager=True 和
|
||||||
|
disable_custom_all_reduce=True 的硬编码,让 Iluvatar 环境可以实际验证
|
||||||
|
CUDA Graph / custom all-reduce;需要回退时仍可通过命令行显式传
|
||||||
|
--enforce-eager --disable-custom-all-reduce。
|
||||||
|
|
||||||
Deploy:
|
Deploy:
|
||||||
python3 modified_scripts/patch_xformers_sdpa_seq.py
|
python3 modified_scripts/patch_xformers_sdpa_seq.py
|
||||||
@@ -91,6 +95,13 @@ _ARG_NEW_BLOCK = """\
|
|||||||
# handles long-context memory without chunked prefill\
|
# handles long-context memory without chunked prefill\
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
_ARG_FORCE_EAGER_OLD = " enforce_eager=True,"
|
||||||
|
_ARG_FORCE_EAGER_NEW = " enforce_eager=self.enforce_eager,"
|
||||||
|
_ARG_FORCE_ALLREDUCE_OLD = " disable_custom_all_reduce=True,"
|
||||||
|
_ARG_FORCE_ALLREDUCE_NEW = (
|
||||||
|
" disable_custom_all_reduce=self.disable_custom_all_reduce,"
|
||||||
|
)
|
||||||
|
|
||||||
FALLBACK_METHOD = '''
|
FALLBACK_METHOD = '''
|
||||||
def _run_sdpa_fallback(
|
def _run_sdpa_fallback(
|
||||||
self,
|
self,
|
||||||
@@ -275,6 +286,26 @@ def patch_arg_utils(path):
|
|||||||
else:
|
else:
|
||||||
print(" [warn] target block not found — check arg_utils.py version")
|
print(" [warn] target block not found — check arg_utils.py version")
|
||||||
|
|
||||||
|
if _ARG_FORCE_EAGER_NEW in content:
|
||||||
|
print(" [skip] enforce_eager already respects CLI")
|
||||||
|
elif _ARG_FORCE_EAGER_OLD in content:
|
||||||
|
content = content.replace(_ARG_FORCE_EAGER_OLD,
|
||||||
|
_ARG_FORCE_EAGER_NEW, 1)
|
||||||
|
print(" [ok] enforce_eager now respects CLI")
|
||||||
|
changed = True
|
||||||
|
else:
|
||||||
|
print(" [warn] enforce_eager assignment not found")
|
||||||
|
|
||||||
|
if _ARG_FORCE_ALLREDUCE_NEW in content:
|
||||||
|
print(" [skip] custom all-reduce already respects CLI")
|
||||||
|
elif _ARG_FORCE_ALLREDUCE_OLD in content:
|
||||||
|
content = content.replace(_ARG_FORCE_ALLREDUCE_OLD,
|
||||||
|
_ARG_FORCE_ALLREDUCE_NEW, 1)
|
||||||
|
print(" [ok] custom all-reduce now respects CLI")
|
||||||
|
changed = True
|
||||||
|
else:
|
||||||
|
print(" [warn] custom all-reduce assignment not found")
|
||||||
|
|
||||||
if changed:
|
if changed:
|
||||||
with open(path, "w") as f:
|
with open(path, "w") as f:
|
||||||
f.write(content)
|
f.write(content)
|
||||||
|
|||||||
@@ -320,12 +320,20 @@ class ChatCompletionRequest(OpenAIBaseModel):
|
|||||||
prompt_logprobs = self.top_logprobs
|
prompt_logprobs = self.top_logprobs
|
||||||
|
|
||||||
guided_json_object = None
|
guided_json_object = None
|
||||||
if (self.response_format is not None
|
guided_json_from_schema = None
|
||||||
and self.response_format.type == "json_object"):
|
if self.response_format is not None:
|
||||||
guided_json_object = True
|
if self.response_format.type == "json_object":
|
||||||
|
guided_json_object = True
|
||||||
|
elif (self.response_format.type == "json_schema"
|
||||||
|
and self.response_format.json_schema is not None
|
||||||
|
and self.response_format.json_schema.json_schema is not None):
|
||||||
|
guided_json_from_schema = \
|
||||||
|
self.response_format.json_schema.json_schema
|
||||||
|
|
||||||
guided_decoding = GuidedDecodingParams.from_optional(
|
guided_decoding = GuidedDecodingParams.from_optional(
|
||||||
json=self._get_guided_json_from_tool() or self.guided_json,
|
json=(self._get_guided_json_from_tool()
|
||||||
|
or self.guided_json
|
||||||
|
or guided_json_from_schema),
|
||||||
regex=self.guided_regex,
|
regex=self.guided_regex,
|
||||||
choice=self.guided_choice,
|
choice=self.guided_choice,
|
||||||
grammar=self.guided_grammar,
|
grammar=self.guided_grammar,
|
||||||
@@ -398,6 +406,10 @@ class ChatCompletionRequest(OpenAIBaseModel):
|
|||||||
normalized.append(msg)
|
normalized.append(msg)
|
||||||
continue
|
continue
|
||||||
if msg.get("content") is None:
|
if msg.get("content") is None:
|
||||||
|
if msg.get("reasoning_content") is None:
|
||||||
|
raise ValueError(
|
||||||
|
"Each message must have at least one of 'content' or "
|
||||||
|
"'reasoning_content'.")
|
||||||
msg = {**msg, "content": ""}
|
msg = {**msg, "content": ""}
|
||||||
normalized.append(msg)
|
normalized.append(msg)
|
||||||
data = {**data, "messages": normalized}
|
data = {**data, "messages": normalized}
|
||||||
@@ -639,12 +651,18 @@ class CompletionRequest(OpenAIBaseModel):
|
|||||||
echo_without_generation = self.echo and self.max_tokens == 0
|
echo_without_generation = self.echo and self.max_tokens == 0
|
||||||
|
|
||||||
guided_json_object = None
|
guided_json_object = None
|
||||||
if (self.response_format is not None
|
guided_json_from_schema = None
|
||||||
and self.response_format.type == "json_object"):
|
if self.response_format is not None:
|
||||||
guided_json_object = True
|
if self.response_format.type == "json_object":
|
||||||
|
guided_json_object = True
|
||||||
|
elif (self.response_format.type == "json_schema"
|
||||||
|
and self.response_format.json_schema is not None
|
||||||
|
and self.response_format.json_schema.json_schema is not None):
|
||||||
|
guided_json_from_schema = \
|
||||||
|
self.response_format.json_schema.json_schema
|
||||||
|
|
||||||
guided_decoding = GuidedDecodingParams.from_optional(
|
guided_decoding = GuidedDecodingParams.from_optional(
|
||||||
json=self.guided_json,
|
json=self.guided_json or guided_json_from_schema,
|
||||||
regex=self.guided_regex,
|
regex=self.guided_regex,
|
||||||
choice=self.guided_choice,
|
choice=self.guided_choice,
|
||||||
grammar=self.guided_grammar,
|
grammar=self.guided_grammar,
|
||||||
|
|||||||
@@ -3,6 +3,9 @@
|
|||||||
# Text-only (no VL, no MTP).
|
# Text-only (no VL, no MTP).
|
||||||
|
|
||||||
from collections import OrderedDict
|
from collections import OrderedDict
|
||||||
|
from contextlib import contextmanager
|
||||||
|
import os
|
||||||
|
import time
|
||||||
from typing import Dict, Iterable, List, Optional, Tuple
|
from typing import Dict, Iterable, List, Optional, Tuple
|
||||||
|
|
||||||
import torch
|
import torch
|
||||||
@@ -41,6 +44,61 @@ from vllm.model_executor.models.interfaces import HasInnerState, SupportsLoRA
|
|||||||
|
|
||||||
logger = init_logger(__name__)
|
logger = init_logger(__name__)
|
||||||
|
|
||||||
|
_ENGINEX_PROFILE_ENABLED = os.getenv("ENGINEX_PROFILE_DECODE", "0") == "1"
|
||||||
|
_ENGINEX_PROFILE_EVERY = int(os.getenv("ENGINEX_PROFILE_EVERY", "32"))
|
||||||
|
_ENGINEX_PROFILE_SYNC = os.getenv("ENGINEX_PROFILE_SYNC", "1") != "0"
|
||||||
|
_ENGINEX_MOE_TINY_IMPL = os.getenv("ENGINEX_MOE_TINY_IMPL", "tokenwise")
|
||||||
|
_ENGINEX_MOE_TINY_MAX = int(os.getenv("ENGINEX_MOE_TINY_MAX", "4"))
|
||||||
|
_enginex_profile_stats: Dict[str, List[float]] = {}
|
||||||
|
_enginex_profile_steps = 0
|
||||||
|
_enginex_profile_mode = "unknown"
|
||||||
|
|
||||||
|
|
||||||
|
def _enginex_profile_active() -> bool:
|
||||||
|
return _ENGINEX_PROFILE_ENABLED and torch.cuda.is_available()
|
||||||
|
|
||||||
|
|
||||||
|
def _enginex_profile_sync() -> None:
|
||||||
|
if _ENGINEX_PROFILE_SYNC:
|
||||||
|
torch.cuda.synchronize()
|
||||||
|
|
||||||
|
|
||||||
|
@contextmanager
|
||||||
|
def _enginex_profile(label: str):
|
||||||
|
if not _enginex_profile_active():
|
||||||
|
yield
|
||||||
|
return
|
||||||
|
label = f"{_enginex_profile_mode}.{label}"
|
||||||
|
_enginex_profile_sync()
|
||||||
|
start = time.perf_counter()
|
||||||
|
try:
|
||||||
|
yield
|
||||||
|
finally:
|
||||||
|
_enginex_profile_sync()
|
||||||
|
elapsed_ms = (time.perf_counter() - start) * 1000.0
|
||||||
|
stat = _enginex_profile_stats.setdefault(label, [0.0, 0.0])
|
||||||
|
stat[0] += elapsed_ms
|
||||||
|
stat[1] += 1.0
|
||||||
|
|
||||||
|
|
||||||
|
def _enginex_profile_log(mode: str) -> None:
|
||||||
|
global _enginex_profile_steps
|
||||||
|
if not _enginex_profile_active():
|
||||||
|
return
|
||||||
|
_enginex_profile_steps += 1
|
||||||
|
if _enginex_profile_steps % max(_ENGINEX_PROFILE_EVERY, 1) != 0:
|
||||||
|
return
|
||||||
|
tp_rank = get_tensor_model_parallel_rank()
|
||||||
|
parts = []
|
||||||
|
for label, (total_ms, count) in sorted(
|
||||||
|
_enginex_profile_stats.items(),
|
||||||
|
key=lambda item: item[1][0],
|
||||||
|
reverse=True):
|
||||||
|
avg_ms = total_ms / max(count, 1.0)
|
||||||
|
parts.append(f"{label}: total={total_ms:.2f}ms avg={avg_ms:.3f}ms n={int(count)}")
|
||||||
|
logger.info("[ENGINEX_PROFILE_QWEN] rank=%d steps=%d mode=%s %s",
|
||||||
|
tp_rank, _enginex_profile_steps, mode, " | ".join(parts))
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# Pure-PyTorch DeltaNet kernels (fallbacks from transformers 5.2.0)
|
# Pure-PyTorch DeltaNet kernels (fallbacks from transformers 5.2.0)
|
||||||
@@ -613,44 +671,48 @@ class Qwen3_5FullAttention(nn.Module):
|
|||||||
total_tokens = hidden_states.shape[0]
|
total_tokens = hidden_states.shape[0]
|
||||||
|
|
||||||
# q_proj output includes gate (dim doubled)
|
# q_proj output includes gate (dim doubled)
|
||||||
qg, _ = self.q_proj(hidden_states) # (total, local_num_heads * head_dim * 2)
|
with _enginex_profile("full_attn.qkv_proj"):
|
||||||
qg = qg.view(total_tokens, self.local_num_heads, self.head_dim * 2)
|
qg, _ = self.q_proj(hidden_states) # (total, local_num_heads * head_dim * 2)
|
||||||
q = qg[:, :, :self.head_dim].reshape(total_tokens, -1)
|
qg = qg.view(total_tokens, self.local_num_heads, self.head_dim * 2)
|
||||||
gate = qg[:, :, self.head_dim:].reshape(total_tokens, -1)
|
q = qg[:, :, :self.head_dim].reshape(total_tokens, -1)
|
||||||
|
gate = qg[:, :, self.head_dim:].reshape(total_tokens, -1)
|
||||||
|
|
||||||
k, _ = self.k_proj(hidden_states) # (total, proj_kv_heads * head_dim)
|
k, _ = self.k_proj(hidden_states) # (total, proj_kv_heads * head_dim)
|
||||||
v, _ = self.v_proj(hidden_states)
|
v, _ = self.v_proj(hidden_states)
|
||||||
|
|
||||||
# q_norm on local Q heads
|
# q_norm on local Q heads
|
||||||
q = self.q_norm.forward_cuda(
|
with _enginex_profile("full_attn.norm_rope"):
|
||||||
q.view(total_tokens, self.local_num_heads, self.head_dim)
|
q = self.q_norm.forward_cuda(
|
||||||
.contiguous()).view(total_tokens, -1)
|
q.view(total_tokens, self.local_num_heads, self.head_dim)
|
||||||
|
.contiguous()).view(total_tokens, -1)
|
||||||
|
|
||||||
# GQA-aware TP: select rank-local KV head BEFORE k_norm and rope so
|
# GQA-aware TP: select rank-local KV head BEFORE k_norm and rope so
|
||||||
# that ixformer kernels always see num_kv_heads=1 (same as 27B path).
|
# that ixformer kernels always see num_kv_heads=1 (same as 27B path).
|
||||||
# Doing k_norm/rope on 2 KV heads (proj_kv_heads=2) triggers ixformer
|
# Doing k_norm/rope on 2 KV heads (proj_kv_heads=2) triggers ixformer
|
||||||
# paths that can produce NaN; restricting to 1 head avoids the issue.
|
# paths that can produce NaN; restricting to 1 head avoids the issue.
|
||||||
if self.q_per_kv_global is not None:
|
if self.q_per_kv_global is not None:
|
||||||
tp_rank = get_tensor_model_parallel_rank()
|
tp_rank = get_tensor_model_parallel_rank()
|
||||||
kv_idx = (tp_rank * self.local_num_heads) // self.q_per_kv_global
|
kv_idx = (tp_rank * self.local_num_heads) // self.q_per_kv_global
|
||||||
k = (k.view(total_tokens, self.proj_kv_heads, self.head_dim)
|
k = (k.view(total_tokens, self.proj_kv_heads, self.head_dim)
|
||||||
[:, kv_idx, :].contiguous()) # (T, head_dim) — 1 head
|
[:, kv_idx, :].contiguous()) # (T, head_dim) — 1 head
|
||||||
v = (v.view(total_tokens, self.proj_kv_heads, self.head_dim)
|
v = (v.view(total_tokens, self.proj_kv_heads, self.head_dim)
|
||||||
[:, kv_idx, :].contiguous()) # (T, head_dim) — 1 head
|
[:, kv_idx, :].contiguous()) # (T, head_dim) — 1 head
|
||||||
|
|
||||||
# k_norm on the (now always 1) rank-local KV head
|
# k_norm on the (now always 1) rank-local KV head
|
||||||
k = self.k_norm.forward_cuda(
|
k = self.k_norm.forward_cuda(
|
||||||
k.view(total_tokens, self.local_num_kv_heads, self.head_dim)
|
k.view(total_tokens, self.local_num_kv_heads, self.head_dim)
|
||||||
.contiguous()).view(total_tokens, -1)
|
.contiguous()).view(total_tokens, -1)
|
||||||
|
|
||||||
# rope: q=(T, local_num_heads*head_dim), k=(T, 1*head_dim) — mirrors 27B
|
# rope: q=(T, local_num_heads*head_dim), k=(T, 1*head_dim) — mirrors 27B
|
||||||
q, k = self.rotary_emb(positions, q, k)
|
q, k = self.rotary_emb(positions, q, k)
|
||||||
|
|
||||||
attn_out = self.attn(q, k, v, kv_cache, attn_metadata)
|
with _enginex_profile("full_attn.paged_attention"):
|
||||||
|
attn_out = self.attn(q, k, v, kv_cache, attn_metadata)
|
||||||
|
|
||||||
# Multiply by sigmoid gate before output projection
|
# Multiply by sigmoid gate before output projection
|
||||||
attn_out = attn_out * torch.sigmoid(gate.float()).to(attn_out.dtype)
|
with _enginex_profile("full_attn.gate_o_proj"):
|
||||||
output, _ = self.o_proj(attn_out)
|
attn_out = attn_out * torch.sigmoid(gate.float()).to(attn_out.dtype)
|
||||||
|
output, _ = self.o_proj(attn_out)
|
||||||
return output
|
return output
|
||||||
|
|
||||||
|
|
||||||
@@ -754,12 +816,13 @@ class Qwen3_5MoeSparseBlock(nn.Module):
|
|||||||
Output is partial (pre-all-reduce), same contract as FusedMoE
|
Output is partial (pre-all-reduce), same contract as FusedMoE
|
||||||
with reduce_results=False.
|
with reduce_results=False.
|
||||||
"""
|
"""
|
||||||
# Routing: softmax → topk → renormalise
|
# Routing: softmax -> topk -> renormalise
|
||||||
routing_weights = torch.softmax(router_logits.float(), dim=-1)
|
with _enginex_profile("moe.routing_topk"):
|
||||||
topk_weights, topk_ids = torch.topk(
|
routing_weights = torch.softmax(router_logits.float(), dim=-1)
|
||||||
routing_weights, self.top_k, dim=-1) # (T, top_k)
|
topk_weights, topk_ids = torch.topk(
|
||||||
topk_weights = topk_weights / topk_weights.sum(dim=-1, keepdim=True)
|
routing_weights, self.top_k, dim=-1) # (T, top_k)
|
||||||
topk_weights = topk_weights.to(hidden_states.dtype)
|
topk_weights = topk_weights / topk_weights.sum(dim=-1, keepdim=True)
|
||||||
|
topk_weights = topk_weights.to(hidden_states.dtype)
|
||||||
|
|
||||||
w13 = self.experts.w13_weight # (E, 2*I, H)
|
w13 = self.experts.w13_weight # (E, 2*I, H)
|
||||||
w2 = self.experts.w2_weight # (E, H, I)
|
w2 = self.experts.w2_weight # (E, H, I)
|
||||||
@@ -771,59 +834,114 @@ class Qwen3_5MoeSparseBlock(nn.Module):
|
|||||||
# gate_up: 1 large GEMM (1,H) × (K*2*I,H)^T → (1, K*2*I)
|
# gate_up: 1 large GEMM (1,H) × (K*2*I,H)^T → (1, K*2*I)
|
||||||
# down: 1 bmm (K,H,I) @ (K,I,1) → (K,H)
|
# down: 1 bmm (K,H,I) @ (K,I,1) → (K,H)
|
||||||
# Total: 3 kernel launches vs previous 16 (top_k*2).
|
# Total: 3 kernel launches vs previous 16 (top_k*2).
|
||||||
eids = topk_ids[0] # (K,)
|
with _enginex_profile("moe.routed_decode_experts"):
|
||||||
ws = topk_weights[0].to(hidden_states.dtype) # (K,)
|
eids = topk_ids[0] # (K,)
|
||||||
w13_sel = w13[eids] # (K, 2*I, H)
|
ws = topk_weights[0].to(hidden_states.dtype) # (K,)
|
||||||
w2_sel = w2[eids] # (K, H, I)
|
w13_sel = w13[eids] # (K, 2*I, H)
|
||||||
|
w2_sel = w2[eids] # (K, H, I)
|
||||||
|
|
||||||
H = hidden_states.shape[-1]
|
H = hidden_states.shape[-1]
|
||||||
|
|
||||||
gate_up = F.linear(
|
gate_up = F.linear(
|
||||||
hidden_states,
|
hidden_states,
|
||||||
w13_sel.reshape(-1, H), # (K*2*I, H) — contiguous after indexing
|
w13_sel.reshape(-1, H), # (K*2*I, H) - contiguous after indexing
|
||||||
) # (1, K*2*I)
|
) # (1, K*2*I)
|
||||||
gate_up = gate_up.view(self.top_k, -1) # (K, 2*I)
|
gate_up = gate_up.view(self.top_k, -1) # (K, 2*I)
|
||||||
gate, up = gate_up.chunk(2, dim=-1) # (K, I) each
|
gate, up = gate_up.chunk(2, dim=-1) # (K, I) each
|
||||||
act = F.silu(gate) * up # (K, I)
|
act = F.silu(gate) * up # (K, I)
|
||||||
|
|
||||||
# bmm: (K,H,I) @ (K,I,1) → (K,H,1) → (K,H)
|
# bmm: (K,H,I) @ (K,I,1) -> (K,H,1) -> (K,H)
|
||||||
expert_out = torch.bmm(w2_sel, act.unsqueeze(-1)).squeeze(-1) # (K, H)
|
expert_out = torch.bmm(w2_sel, act.unsqueeze(-1)).squeeze(-1) # (K, H)
|
||||||
|
|
||||||
out = (expert_out * ws.unsqueeze(-1)).sum(0, keepdim=True).to(
|
out = (expert_out * ws.unsqueeze(-1)).sum(0, keepdim=True).to(
|
||||||
hidden_states.dtype) # (1, H)
|
hidden_states.dtype) # (1, H)
|
||||||
|
elif T <= _ENGINEX_MOE_TINY_MAX and _ENGINEX_MOE_TINY_IMPL == "tokenwise":
|
||||||
|
# Fast path: tiny decode batch. Compute each token with the proven
|
||||||
|
# T==1 large-F.linear path, avoiding the per-expert Python loop.
|
||||||
|
# This usually beats batched bmm for very small T because the first
|
||||||
|
# projection becomes T larger GEMMs instead of T*K tiny GEMMs.
|
||||||
|
with _enginex_profile("moe.routed_tiny_tokenwise_experts"):
|
||||||
|
H = hidden_states.shape[-1]
|
||||||
|
pieces = []
|
||||||
|
for t in range(T):
|
||||||
|
eids = topk_ids[t]
|
||||||
|
ws = topk_weights[t].to(hidden_states.dtype)
|
||||||
|
w13_sel = w13[eids] # (K, 2*I, H)
|
||||||
|
w2_sel = w2[eids] # (K, H, I)
|
||||||
|
|
||||||
|
gate_up = F.linear(
|
||||||
|
hidden_states[t:t + 1],
|
||||||
|
w13_sel.reshape(-1, H),
|
||||||
|
).view(self.top_k, -1) # (K, 2*I)
|
||||||
|
gate, up = gate_up.chunk(2, dim=-1)
|
||||||
|
act = F.silu(gate) * up
|
||||||
|
expert_out = torch.bmm(
|
||||||
|
w2_sel, act.unsqueeze(-1)).squeeze(-1) # (K, H)
|
||||||
|
pieces.append((expert_out * ws.unsqueeze(-1)).sum(0))
|
||||||
|
|
||||||
|
out = torch.stack(pieces, dim=0).to(hidden_states.dtype)
|
||||||
|
elif T <= _ENGINEX_MOE_TINY_MAX:
|
||||||
|
# Alternative tiny decode batch implementation. Kept for A/B
|
||||||
|
# testing via ENGINEX_MOE_TINY_IMPL=bmm.
|
||||||
|
with _enginex_profile("moe.routed_tiny_bmm_experts"):
|
||||||
|
H = hidden_states.shape[-1]
|
||||||
|
flat_eids = topk_ids.reshape(-1) # (T*K,)
|
||||||
|
flat_ws = topk_weights.reshape(-1).to(hidden_states.dtype)
|
||||||
|
|
||||||
|
w13_sel = w13[flat_eids] # (T*K, 2*I, H)
|
||||||
|
w2_sel = w2[flat_eids] # (T*K, H, I)
|
||||||
|
x = (hidden_states[:, None, :]
|
||||||
|
.expand(T, self.top_k, H)
|
||||||
|
.reshape(-1, 1, H)) # (T*K, 1, H)
|
||||||
|
|
||||||
|
gate_up = torch.bmm(
|
||||||
|
x, w13_sel.transpose(1, 2)).squeeze(1) # (T*K, 2*I)
|
||||||
|
gate, up = gate_up.chunk(2, dim=-1)
|
||||||
|
act = F.silu(gate) * up # (T*K, I)
|
||||||
|
expert_out = torch.bmm(
|
||||||
|
w2_sel, act.unsqueeze(-1)).squeeze(-1) # (T*K, H)
|
||||||
|
|
||||||
|
out = (expert_out * flat_ws.unsqueeze(-1)).view(
|
||||||
|
T, self.top_k, H).sum(1).to(hidden_states.dtype)
|
||||||
else:
|
else:
|
||||||
# General path (prefill / multi-seq): loop over unique active experts.
|
# General path (prefill / multi-seq): loop over unique active experts.
|
||||||
# At most T*top_k unique experts, always <= num_experts.
|
# At most T*top_k unique experts, always <= num_experts.
|
||||||
out = torch.zeros_like(hidden_states)
|
with _enginex_profile("moe.routed_prefill_experts"):
|
||||||
unique_eids = topk_ids.view(-1).unique().tolist()
|
out = torch.zeros_like(hidden_states)
|
||||||
for eid in unique_eids:
|
unique_eids = topk_ids.view(-1).unique().tolist()
|
||||||
eid = int(eid)
|
for eid in unique_eids:
|
||||||
mask = (topk_ids == eid) # (T, top_k)
|
eid = int(eid)
|
||||||
tok_ids, topk_pos = mask.nonzero(as_tuple=True)
|
mask = (topk_ids == eid) # (T, top_k)
|
||||||
tokens = hidden_states[tok_ids] # (n, H)
|
tok_ids, topk_pos = mask.nonzero(as_tuple=True)
|
||||||
gate_up = F.linear(tokens, w13[eid]) # (n, 2*I)
|
tokens = hidden_states[tok_ids] # (n, H)
|
||||||
gate, up = gate_up.chunk(2, dim=-1)
|
gate_up = F.linear(tokens, w13[eid]) # (n, 2*I)
|
||||||
act = F.silu(gate) * up # (n, I)
|
gate, up = gate_up.chunk(2, dim=-1)
|
||||||
expert_out = F.linear(act, w2[eid]) # (n, H)
|
act = F.silu(gate) * up # (n, I)
|
||||||
weights = topk_weights[tok_ids, topk_pos].unsqueeze(-1)
|
expert_out = F.linear(act, w2[eid]) # (n, H)
|
||||||
out.index_add_(0, tok_ids, (expert_out * weights).to(out.dtype))
|
weights = topk_weights[tok_ids, topk_pos].unsqueeze(-1)
|
||||||
|
out.index_add_(0, tok_ids, (expert_out * weights).to(out.dtype))
|
||||||
|
|
||||||
return out # partial, all-reduce done in forward()
|
return out # partial, all-reduce done in forward()
|
||||||
|
|
||||||
def forward(self, hidden_states: torch.Tensor) -> torch.Tensor:
|
def forward(self, hidden_states: torch.Tensor) -> torch.Tensor:
|
||||||
router_logits, _ = self.gate(hidden_states)
|
with _enginex_profile("moe.gate"):
|
||||||
routed_out = self._pure_pytorch_experts(hidden_states, router_logits)
|
router_logits, _ = self.gate(hidden_states)
|
||||||
|
with _enginex_profile("moe.routed_total"):
|
||||||
|
routed_out = self._pure_pytorch_experts(hidden_states, router_logits)
|
||||||
|
|
||||||
gate_up, _ = self.shared_expert_gate_up(hidden_states)
|
with _enginex_profile("moe.shared_expert"):
|
||||||
shared_out = self.act_fn(gate_up)
|
gate_up, _ = self.shared_expert_gate_up(hidden_states)
|
||||||
shared_out, _ = self.shared_expert_down(shared_out)
|
shared_out = self.act_fn(gate_up)
|
||||||
# Scalar sigmoid gate (Qwen2-MoE / Qwen3.5-MoE style)
|
shared_out, _ = self.shared_expert_down(shared_out)
|
||||||
gate_score, _ = self.shared_expert_gate(hidden_states) # (T, 1)
|
# Scalar sigmoid gate (Qwen2-MoE / Qwen3.5-MoE style)
|
||||||
shared_out = shared_out * torch.sigmoid(gate_score)
|
gate_score, _ = self.shared_expert_gate(hidden_states) # (T, 1)
|
||||||
|
shared_out = shared_out * torch.sigmoid(gate_score)
|
||||||
|
|
||||||
out = routed_out + shared_out
|
with _enginex_profile("moe.combine"):
|
||||||
|
out = routed_out + shared_out
|
||||||
if self.experts.tp_size > 1:
|
if self.experts.tp_size > 1:
|
||||||
out = tensor_model_parallel_all_reduce(out)
|
with _enginex_profile("moe.tp_all_reduce"):
|
||||||
|
out = tensor_model_parallel_all_reduce(out)
|
||||||
return out
|
return out
|
||||||
|
|
||||||
|
|
||||||
@@ -883,21 +1001,27 @@ class Qwen3_5DecoderLayer(nn.Module):
|
|||||||
) -> Tuple[torch.Tensor, torch.Tensor]:
|
) -> Tuple[torch.Tensor, torch.Tensor]:
|
||||||
if residual is None:
|
if residual is None:
|
||||||
residual = hidden_states
|
residual = hidden_states
|
||||||
hidden_states = self.input_layernorm(hidden_states)
|
with _enginex_profile("layer.input_norm"):
|
||||||
|
hidden_states = self.input_layernorm(hidden_states)
|
||||||
else:
|
else:
|
||||||
hidden_states, residual = self.input_layernorm(hidden_states, residual)
|
with _enginex_profile("layer.input_norm"):
|
||||||
|
hidden_states, residual = self.input_layernorm(hidden_states, residual)
|
||||||
|
|
||||||
if self.layer_type == "linear_attention":
|
if self.layer_type == "linear_attention":
|
||||||
hidden_states = self.linear_attn(
|
with _enginex_profile("layer.linear_attention"):
|
||||||
hidden_states, attn_metadata, conv_state, temporal_state)
|
hidden_states = self.linear_attn(
|
||||||
|
hidden_states, attn_metadata, conv_state, temporal_state)
|
||||||
else:
|
else:
|
||||||
hidden_states = self.self_attn(
|
with _enginex_profile("layer.full_attention"):
|
||||||
positions, hidden_states, kv_cache, attn_metadata)
|
hidden_states = self.self_attn(
|
||||||
|
positions, hidden_states, kv_cache, attn_metadata)
|
||||||
|
|
||||||
hidden_states, residual = self.post_attention_layernorm(
|
with _enginex_profile("layer.post_attn_norm"):
|
||||||
hidden_states, residual)
|
hidden_states, residual = self.post_attention_layernorm(
|
||||||
|
hidden_states, residual)
|
||||||
|
|
||||||
hidden_states = self.mlp(hidden_states)
|
with _enginex_profile("layer.mlp"):
|
||||||
|
hidden_states = self.mlp(hidden_states)
|
||||||
|
|
||||||
return hidden_states, residual
|
return hidden_states, residual
|
||||||
|
|
||||||
@@ -934,6 +1058,9 @@ class Qwen3_5Model(nn.Module):
|
|||||||
conv_states: torch.Tensor, # (num_linear_layers, batch, ...)
|
conv_states: torch.Tensor, # (num_linear_layers, batch, ...)
|
||||||
temporal_states: torch.Tensor, # (num_linear_layers, batch, ...)
|
temporal_states: torch.Tensor, # (num_linear_layers, batch, ...)
|
||||||
) -> torch.Tensor:
|
) -> torch.Tensor:
|
||||||
|
global _enginex_profile_mode
|
||||||
|
mode = "prefill" if attn_metadata.num_prefill_tokens > 0 else "decode"
|
||||||
|
_enginex_profile_mode = mode
|
||||||
hidden_states = self.embed_tokens(input_ids)
|
hidden_states = self.embed_tokens(input_ids)
|
||||||
residual = None
|
residual = None
|
||||||
|
|
||||||
@@ -961,6 +1088,7 @@ class Qwen3_5Model(nn.Module):
|
|||||||
attn_idx += 1
|
attn_idx += 1
|
||||||
|
|
||||||
hidden_states, _ = self.norm(hidden_states, residual)
|
hidden_states, _ = self.norm(hidden_states, residual)
|
||||||
|
_enginex_profile_log(mode)
|
||||||
return hidden_states
|
return hidden_states
|
||||||
|
|
||||||
|
|
||||||
@@ -1322,6 +1450,35 @@ class Qwen3_5MoeForCausalLM(Qwen3_5ForCausalLM):
|
|||||||
weight_loader(param, loaded_weight)
|
weight_loader(param, loaded_weight)
|
||||||
continue
|
continue
|
||||||
|
|
||||||
|
# --- Individual expert weights (FT checkpoint: experts.{i}.{proj}.weight) ---
|
||||||
|
# Standard transformers fine-tuning saves each expert separately instead of
|
||||||
|
# the pre-merged (num_experts, ...) tensors in the original checkpoint.
|
||||||
|
if ".mlp.experts." in name:
|
||||||
|
parts = name.split(".mlp.experts.", 1)
|
||||||
|
expert_rest = parts[1] # e.g. "0.gate_proj.weight"
|
||||||
|
dot_pos = expert_rest.find(".")
|
||||||
|
if dot_pos > 0 and expert_rest[:dot_pos].isdigit():
|
||||||
|
eid = int(expert_rest[:dot_pos])
|
||||||
|
proj_raw = expert_rest[dot_pos + 1:]
|
||||||
|
proj = proj_raw[:-7] if proj_raw.endswith(".weight") else proj_raw
|
||||||
|
prefix = parts[0] # e.g. "model.layers.0"
|
||||||
|
if proj == "gate_proj":
|
||||||
|
w13_name = f"{prefix}.mlp.experts.w13_weight"
|
||||||
|
if w13_name in params_dict:
|
||||||
|
param = params_dict[w13_name]
|
||||||
|
param.weight_loader(param, loaded_weight, "w1_weight", "w1", eid)
|
||||||
|
elif proj == "up_proj":
|
||||||
|
w13_name = f"{prefix}.mlp.experts.w13_weight"
|
||||||
|
if w13_name in params_dict:
|
||||||
|
param = params_dict[w13_name]
|
||||||
|
param.weight_loader(param, loaded_weight, "w3_weight", "w3", eid)
|
||||||
|
elif proj == "down_proj":
|
||||||
|
w2_name = f"{prefix}.mlp.experts.w2_weight"
|
||||||
|
if w2_name in params_dict:
|
||||||
|
param = params_dict[w2_name]
|
||||||
|
param.weight_loader(param, loaded_weight, "w2_weight", "w2", eid)
|
||||||
|
continue
|
||||||
|
|
||||||
# --- Stacked / standard weights ---
|
# --- Stacked / standard weights ---
|
||||||
for param_name, weight_name, shard_id in stacked_params_mapping:
|
for param_name, weight_name, shard_id in stacked_params_mapping:
|
||||||
if weight_name not in name:
|
if weight_name not in name:
|
||||||
|
|||||||
@@ -294,12 +294,12 @@ class OpenAIServingChat(OpenAIServing):
|
|||||||
if request.stream:
|
if request.stream:
|
||||||
return self.chat_completion_stream_generator(
|
return self.chat_completion_stream_generator(
|
||||||
request, result_generator, request_id, conversation, tokenizer,
|
request, result_generator, request_id, conversation, tokenizer,
|
||||||
request_metadata)
|
request_metadata, raw_request=raw_request)
|
||||||
|
|
||||||
try:
|
try:
|
||||||
return await self.chat_completion_full_generator(
|
return await self.chat_completion_full_generator(
|
||||||
request, result_generator, request_id, conversation, tokenizer,
|
request, result_generator, request_id, conversation, tokenizer,
|
||||||
request_metadata)
|
request_metadata, raw_request=raw_request)
|
||||||
except ValueError as e:
|
except ValueError as e:
|
||||||
# TODO: Use a vllm-specific Validation Error
|
# TODO: Use a vllm-specific Validation Error
|
||||||
return self.create_error_response(str(e))
|
return self.create_error_response(str(e))
|
||||||
@@ -317,6 +317,7 @@ class OpenAIServingChat(OpenAIServing):
|
|||||||
conversation: List[ConversationMessage],
|
conversation: List[ConversationMessage],
|
||||||
tokenizer: AnyTokenizer,
|
tokenizer: AnyTokenizer,
|
||||||
request_metadata: RequestResponseMetadata,
|
request_metadata: RequestResponseMetadata,
|
||||||
|
raw_request: Optional[Request] = None,
|
||||||
) -> AsyncGenerator[str, None]:
|
) -> AsyncGenerator[str, None]:
|
||||||
model_name = self.base_model_paths[0].name
|
model_name = self.base_model_paths[0].name
|
||||||
created_time = int(time.time())
|
created_time = int(time.time())
|
||||||
@@ -390,6 +391,27 @@ class OpenAIServingChat(OpenAIServing):
|
|||||||
yield "data: [DONE]\n\n"
|
yield "data: [DONE]\n\n"
|
||||||
return
|
return
|
||||||
|
|
||||||
|
# Background task: poll is_disconnected() every 300 ms and abort the
|
||||||
|
# engine request as soon as the client goes away. This catches the
|
||||||
|
# case where the HTTP layer (Starlette/uvicorn) does not actively read
|
||||||
|
# the receive channel during streaming, so is_disconnected() in
|
||||||
|
# iterate_with_cancellation never fires during fast decode.
|
||||||
|
_disconnect_watcher: Optional[asyncio.Task] = None
|
||||||
|
if raw_request is not None:
|
||||||
|
async def _watch_disconnect() -> None:
|
||||||
|
try:
|
||||||
|
while True:
|
||||||
|
if await raw_request.is_disconnected():
|
||||||
|
logger.info(
|
||||||
|
"Client disconnected (decode watcher), "
|
||||||
|
"aborting request %s", request_id)
|
||||||
|
await self.engine_client.abort(request_id)
|
||||||
|
return
|
||||||
|
await asyncio.sleep(0.3)
|
||||||
|
except asyncio.CancelledError:
|
||||||
|
pass
|
||||||
|
_disconnect_watcher = asyncio.ensure_future(_watch_disconnect())
|
||||||
|
|
||||||
try:
|
try:
|
||||||
async for res in result_generator:
|
async for res in result_generator:
|
||||||
if res.prompt_token_ids is not None:
|
if res.prompt_token_ids is not None:
|
||||||
@@ -732,7 +754,7 @@ class OpenAIServingChat(OpenAIServing):
|
|||||||
reasoning_tokens=total_reasoning)
|
reasoning_tokens=total_reasoning)
|
||||||
|
|
||||||
except asyncio.CancelledError:
|
except asyncio.CancelledError:
|
||||||
# Client disconnected; abort the engine request so GPU is freed.
|
# Client disconnected via CancelledError path; abort engine request.
|
||||||
await self.engine_client.abort(request_id)
|
await self.engine_client.abort(request_id)
|
||||||
return
|
return
|
||||||
except ValueError as e:
|
except ValueError as e:
|
||||||
@@ -740,6 +762,18 @@ class OpenAIServingChat(OpenAIServing):
|
|||||||
logger.error("error in chat completion stream generator: %s", e)
|
logger.error("error in chat completion stream generator: %s", e)
|
||||||
data = self.create_streaming_error_response(str(e))
|
data = self.create_streaming_error_response(str(e))
|
||||||
yield f"data: {data}\n\n"
|
yield f"data: {data}\n\n"
|
||||||
|
finally:
|
||||||
|
# Stop the disconnect watcher (it may already be done if it fired).
|
||||||
|
if _disconnect_watcher is not None and not _disconnect_watcher.done():
|
||||||
|
_disconnect_watcher.cancel()
|
||||||
|
try:
|
||||||
|
await _disconnect_watcher
|
||||||
|
except asyncio.CancelledError:
|
||||||
|
pass
|
||||||
|
# Covers GeneratorExit when Starlette calls aclose() on disconnect
|
||||||
|
# during decode (tokens arrive fast so CancelledError path is not
|
||||||
|
# always triggered). abort() is a no-op for already-finished requests.
|
||||||
|
await self.engine_client.abort(request_id)
|
||||||
# Send the final done message after all response.n are finished
|
# Send the final done message after all response.n are finished
|
||||||
yield "data: [DONE]\n\n"
|
yield "data: [DONE]\n\n"
|
||||||
|
|
||||||
@@ -751,18 +785,47 @@ class OpenAIServingChat(OpenAIServing):
|
|||||||
conversation: List[ConversationMessage],
|
conversation: List[ConversationMessage],
|
||||||
tokenizer: AnyTokenizer,
|
tokenizer: AnyTokenizer,
|
||||||
request_metadata: RequestResponseMetadata,
|
request_metadata: RequestResponseMetadata,
|
||||||
|
raw_request: Optional[Request] = None,
|
||||||
) -> Union[ErrorResponse, ChatCompletionResponse]:
|
) -> Union[ErrorResponse, ChatCompletionResponse]:
|
||||||
|
|
||||||
model_name = self.base_model_paths[0].name
|
model_name = self.base_model_paths[0].name
|
||||||
created_time = int(time.time())
|
created_time = int(time.time())
|
||||||
final_res: Optional[RequestOutput] = None
|
final_res: Optional[RequestOutput] = None
|
||||||
|
|
||||||
|
# Background watcher: same logic as the streaming path — polls
|
||||||
|
# is_disconnected() every 300 ms so that a client disconnect during
|
||||||
|
# non-streaming decode is caught even when uvicorn isn't actively
|
||||||
|
# reading the receive channel.
|
||||||
|
_disconnect_watcher: Optional[asyncio.Task] = None
|
||||||
|
if raw_request is not None:
|
||||||
|
async def _watch_disconnect() -> None:
|
||||||
|
try:
|
||||||
|
while True:
|
||||||
|
if await raw_request.is_disconnected():
|
||||||
|
logger.info(
|
||||||
|
"Client disconnected (non-stream watcher), "
|
||||||
|
"aborting request %s", request_id)
|
||||||
|
await self.engine_client.abort(request_id)
|
||||||
|
return
|
||||||
|
await asyncio.sleep(0.3)
|
||||||
|
except asyncio.CancelledError:
|
||||||
|
pass
|
||||||
|
_disconnect_watcher = asyncio.ensure_future(_watch_disconnect())
|
||||||
|
|
||||||
try:
|
try:
|
||||||
async for res in result_generator:
|
async for res in result_generator:
|
||||||
final_res = res
|
final_res = res
|
||||||
except asyncio.CancelledError:
|
except asyncio.CancelledError:
|
||||||
await self.engine_client.abort(request_id)
|
await self.engine_client.abort(request_id)
|
||||||
return self.create_error_response("Client disconnected")
|
return self.create_error_response("Client disconnected")
|
||||||
|
finally:
|
||||||
|
if _disconnect_watcher is not None and not _disconnect_watcher.done():
|
||||||
|
_disconnect_watcher.cancel()
|
||||||
|
try:
|
||||||
|
await _disconnect_watcher
|
||||||
|
except asyncio.CancelledError:
|
||||||
|
pass
|
||||||
|
await self.engine_client.abort(request_id)
|
||||||
|
|
||||||
assert final_res is not None
|
assert final_res is not None
|
||||||
|
|
||||||
|
|||||||
@@ -851,7 +851,7 @@ class EngineArgs:
|
|||||||
max_model_len=self.max_model_len,
|
max_model_len=self.max_model_len,
|
||||||
quantization=self.quantization,
|
quantization=self.quantization,
|
||||||
quantization_param_path=self.quantization_param_path,
|
quantization_param_path=self.quantization_param_path,
|
||||||
enforce_eager=True,
|
enforce_eager=self.enforce_eager,
|
||||||
max_context_len_to_capture=self.max_context_len_to_capture,
|
max_context_len_to_capture=self.max_context_len_to_capture,
|
||||||
max_seq_len_to_capture=self.max_seq_len_to_capture,
|
max_seq_len_to_capture=self.max_seq_len_to_capture,
|
||||||
max_logprobs=self.max_logprobs,
|
max_logprobs=self.max_logprobs,
|
||||||
@@ -927,7 +927,7 @@ class EngineArgs:
|
|||||||
tensor_parallel_size=self.tensor_parallel_size,
|
tensor_parallel_size=self.tensor_parallel_size,
|
||||||
worker_use_ray=self.worker_use_ray,
|
worker_use_ray=self.worker_use_ray,
|
||||||
max_parallel_loading_workers=self.max_parallel_loading_workers,
|
max_parallel_loading_workers=self.max_parallel_loading_workers,
|
||||||
disable_custom_all_reduce=True,
|
disable_custom_all_reduce=self.disable_custom_all_reduce,
|
||||||
tokenizer_pool_config=TokenizerPoolConfig.create_config(
|
tokenizer_pool_config=TokenizerPoolConfig.create_config(
|
||||||
self.tokenizer_pool_size,
|
self.tokenizer_pool_size,
|
||||||
self.tokenizer_pool_type,
|
self.tokenizer_pool_type,
|
||||||
|
|||||||
@@ -2,9 +2,11 @@ import dataclasses
|
|||||||
import gc
|
import gc
|
||||||
import inspect
|
import inspect
|
||||||
import itertools
|
import itertools
|
||||||
|
import os
|
||||||
import time
|
import time
|
||||||
import warnings
|
import warnings
|
||||||
import weakref
|
import weakref
|
||||||
|
from contextlib import contextmanager
|
||||||
from dataclasses import dataclass
|
from dataclasses import dataclass
|
||||||
from typing import (TYPE_CHECKING, Any, Callable, Dict, List, Optional, Set,
|
from typing import (TYPE_CHECKING, Any, Callable, Dict, List, Optional, Set,
|
||||||
Tuple, Type, TypeVar, Union)
|
Tuple, Type, TypeVar, Union)
|
||||||
@@ -62,6 +64,56 @@ if TYPE_CHECKING:
|
|||||||
|
|
||||||
logger = init_logger(__name__)
|
logger = init_logger(__name__)
|
||||||
|
|
||||||
|
_ENGINEX_PROFILE_ENABLED = os.getenv("ENGINEX_PROFILE_DECODE", "0") == "1"
|
||||||
|
_ENGINEX_PROFILE_EVERY = int(os.getenv("ENGINEX_PROFILE_EVERY", "32"))
|
||||||
|
_ENGINEX_PROFILE_SYNC = os.getenv("ENGINEX_PROFILE_SYNC", "1") != "0"
|
||||||
|
_enginex_profile_stats: Dict[str, List[float]] = {}
|
||||||
|
_enginex_profile_steps = 0
|
||||||
|
|
||||||
|
|
||||||
|
def _enginex_profile_active() -> bool:
|
||||||
|
return _ENGINEX_PROFILE_ENABLED and torch.cuda.is_available()
|
||||||
|
|
||||||
|
|
||||||
|
def _enginex_profile_sync() -> None:
|
||||||
|
if _ENGINEX_PROFILE_SYNC:
|
||||||
|
torch.cuda.synchronize()
|
||||||
|
|
||||||
|
|
||||||
|
@contextmanager
|
||||||
|
def _enginex_profile(label: str):
|
||||||
|
if not _enginex_profile_active():
|
||||||
|
yield
|
||||||
|
return
|
||||||
|
_enginex_profile_sync()
|
||||||
|
start = time.perf_counter()
|
||||||
|
try:
|
||||||
|
yield
|
||||||
|
finally:
|
||||||
|
_enginex_profile_sync()
|
||||||
|
elapsed_ms = (time.perf_counter() - start) * 1000.0
|
||||||
|
stat = _enginex_profile_stats.setdefault(label, [0.0, 0.0])
|
||||||
|
stat[0] += elapsed_ms
|
||||||
|
stat[1] += 1.0
|
||||||
|
|
||||||
|
|
||||||
|
def _enginex_profile_log(mode: Optional[str]) -> None:
|
||||||
|
global _enginex_profile_steps
|
||||||
|
if not _enginex_profile_active():
|
||||||
|
return
|
||||||
|
_enginex_profile_steps += 1
|
||||||
|
if _enginex_profile_steps % max(_ENGINEX_PROFILE_EVERY, 1) != 0:
|
||||||
|
return
|
||||||
|
parts = []
|
||||||
|
for label, (total_ms, count) in sorted(
|
||||||
|
_enginex_profile_stats.items(),
|
||||||
|
key=lambda item: item[1][0],
|
||||||
|
reverse=True):
|
||||||
|
avg_ms = total_ms / max(count, 1.0)
|
||||||
|
parts.append(f"{label}: total={total_ms:.2f}ms avg={avg_ms:.3f}ms n={int(count)}")
|
||||||
|
logger.info("[ENGINEX_PROFILE_MODEL_RUNNER] steps=%d mode=%s %s",
|
||||||
|
_enginex_profile_steps, mode, " | ".join(parts))
|
||||||
|
|
||||||
LORA_WARMUP_RANK = 8
|
LORA_WARMUP_RANK = 8
|
||||||
_BATCH_SIZE_ALIGNMENT = 8
|
_BATCH_SIZE_ALIGNMENT = 8
|
||||||
# all the token sizes that **can** be captured by cudagraph.
|
# all the token sizes that **can** be captured by cudagraph.
|
||||||
@@ -1633,7 +1685,9 @@ class ModelRunner(GPUModelRunnerBase[ModelInputForGPUWithSamplingMetadata]):
|
|||||||
model_input.prompt_adapter_requests,
|
model_input.prompt_adapter_requests,
|
||||||
model_input.prompt_adapter_mapping)
|
model_input.prompt_adapter_mapping)
|
||||||
|
|
||||||
self.attn_state.begin_forward(model_input)
|
profile_mode = "prompt" if model_input.is_prompt else "decode"
|
||||||
|
with _enginex_profile(f"{profile_mode}.attn_begin_forward"):
|
||||||
|
self.attn_state.begin_forward(model_input)
|
||||||
|
|
||||||
# Currently cuda graph is only supported by the decode phase.
|
# Currently cuda graph is only supported by the decode phase.
|
||||||
assert model_input.attn_metadata is not None
|
assert model_input.attn_metadata is not None
|
||||||
@@ -1661,16 +1715,17 @@ class ModelRunner(GPUModelRunnerBase[ModelInputForGPUWithSamplingMetadata]):
|
|||||||
model_forward_end = torch.cuda.Event(enable_timing=True)
|
model_forward_end = torch.cuda.Event(enable_timing=True)
|
||||||
model_forward_start.record()
|
model_forward_start.record()
|
||||||
|
|
||||||
with set_forward_context(model_input.attn_metadata):
|
with _enginex_profile(f"{profile_mode}.model_forward"):
|
||||||
hidden_or_intermediate_states = model_executable(
|
with set_forward_context(model_input.attn_metadata):
|
||||||
input_ids=model_input.input_tokens,
|
hidden_or_intermediate_states = model_executable(
|
||||||
positions=model_input.input_positions,
|
input_ids=model_input.input_tokens,
|
||||||
kv_caches=kv_caches,
|
positions=model_input.input_positions,
|
||||||
attn_metadata=model_input.attn_metadata,
|
kv_caches=kv_caches,
|
||||||
intermediate_tensors=intermediate_tensors,
|
attn_metadata=model_input.attn_metadata,
|
||||||
**MultiModalInputs.as_kwargs(multi_modal_kwargs,
|
intermediate_tensors=intermediate_tensors,
|
||||||
device=self.device),
|
**MultiModalInputs.as_kwargs(multi_modal_kwargs,
|
||||||
**seqlen_agnostic_kwargs)
|
device=self.device),
|
||||||
|
**seqlen_agnostic_kwargs)
|
||||||
|
|
||||||
if (self.observability_config is not None
|
if (self.observability_config is not None
|
||||||
and self.observability_config.collect_model_forward_time):
|
and self.observability_config.collect_model_forward_time):
|
||||||
@@ -1695,20 +1750,23 @@ class ModelRunner(GPUModelRunnerBase[ModelInputForGPUWithSamplingMetadata]):
|
|||||||
torch.tensor(model_forward_time + orig_model_forward_time))
|
torch.tensor(model_forward_time + orig_model_forward_time))
|
||||||
return hidden_or_intermediate_states
|
return hidden_or_intermediate_states
|
||||||
|
|
||||||
logits = self.model.compute_logits(hidden_or_intermediate_states,
|
with _enginex_profile(f"{profile_mode}.compute_logits"):
|
||||||
model_input.sampling_metadata)
|
logits = self.model.compute_logits(hidden_or_intermediate_states,
|
||||||
|
model_input.sampling_metadata)
|
||||||
|
|
||||||
if not self.is_driver_worker:
|
if not self.is_driver_worker:
|
||||||
return []
|
return []
|
||||||
|
|
||||||
if model_input.async_callback is not None:
|
if model_input.async_callback is not None:
|
||||||
model_input.async_callback()
|
with _enginex_profile(f"{profile_mode}.async_callback"):
|
||||||
|
model_input.async_callback()
|
||||||
|
|
||||||
# Sample the next token.
|
# Sample the next token.
|
||||||
output: SamplerOutput = self.model.sample(
|
with _enginex_profile(f"{profile_mode}.sample"):
|
||||||
logits=logits,
|
output: SamplerOutput = self.model.sample(
|
||||||
sampling_metadata=model_input.sampling_metadata,
|
logits=logits,
|
||||||
)
|
sampling_metadata=model_input.sampling_metadata,
|
||||||
|
)
|
||||||
if (self.observability_config is not None
|
if (self.observability_config is not None
|
||||||
and self.observability_config.collect_model_forward_time
|
and self.observability_config.collect_model_forward_time
|
||||||
and output is not None):
|
and output is not None):
|
||||||
@@ -1741,6 +1799,7 @@ class ModelRunner(GPUModelRunnerBase[ModelInputForGPUWithSamplingMetadata]):
|
|||||||
|
|
||||||
output.hidden_states = hidden_states
|
output.hidden_states = hidden_states
|
||||||
|
|
||||||
|
_enginex_profile_log(profile_mode)
|
||||||
return [output]
|
return [output]
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
289
worklogs/2026-07-13-initial-run.md
Normal file
289
worklogs/2026-07-13-initial-run.md
Normal file
@@ -0,0 +1,289 @@
|
|||||||
|
# 2026-07-13 Initial Run Worklog
|
||||||
|
|
||||||
|
## Working Rules
|
||||||
|
|
||||||
|
- Local repository is the source of truth for code changes.
|
||||||
|
- Remote Phanthy GPU server is used for build, runtime, and validation.
|
||||||
|
- Every meaningful experiment records: code version, command, environment, result, issue, and next action.
|
||||||
|
- Changes should be committed with git after a coherent milestone or before risky experiments.
|
||||||
|
|
||||||
|
## Remote Target
|
||||||
|
|
||||||
|
- Host: `ssh-55c3b0b3.default.gpu.phanthy.com`
|
||||||
|
- SSH requires TLS ProxyCommand on port `32222`.
|
||||||
|
- User: `root`
|
||||||
|
|
||||||
|
## Status
|
||||||
|
|
||||||
|
- Local repository inspected.
|
||||||
|
- SSH connectivity confirmed with `whoami` and `hostname`.
|
||||||
|
- Remote environment inspected.
|
||||||
|
|
||||||
|
## Remote Environment Findings
|
||||||
|
|
||||||
|
- Remote shell user: `root`
|
||||||
|
- Remote hostname: `cc-55c3b0b3-8c03-4fe2-8ef8-7109a6aff0d6-0`
|
||||||
|
- Remote appears to already be inside a container.
|
||||||
|
- `docker` is not installed in the remote runtime container.
|
||||||
|
- CoreX is installed under `/usr/local/corex -> /usr/local/corex-3.2.3`.
|
||||||
|
- Iluvatar devices are visible as `/dev/iluvatar0` through `/dev/iluvatar3`.
|
||||||
|
- PyTorch is available only when `PYTHONPATH` and `LD_LIBRARY_PATH` include CoreX paths.
|
||||||
|
- PyTorch reports `torch.cuda.is_available() == True` and `torch.cuda.device_count() == 4`.
|
||||||
|
- Model path found: `/root/public-storage/models/Qwen/Qwen3.6-35B-A3B`.
|
||||||
|
- Model `config.json` already has `architectures: ["Qwen3_5MoeForCausalLM"]`.
|
||||||
|
- Because remote has no Docker, first run will patch the current runtime directly instead of building an image.
|
||||||
|
|
||||||
|
## First Run Plan
|
||||||
|
|
||||||
|
1. Sync only lightweight submission files to `/root/work/enginex-vllm-bi100-qwen36`.
|
||||||
|
2. Backup target runtime files before applying `qwen3_6_scripts/patch_ops.sh`.
|
||||||
|
3. Apply patches in the remote CoreX container.
|
||||||
|
4. Start OpenAI-compatible API server against `/root/public-storage/models/Qwen/Qwen3.6-35B-A3B`.
|
||||||
|
5. Run minimal smoke tests before deeper compatibility tests.
|
||||||
|
|
||||||
|
## 2026-07-14 First Patch Attempt Diagnosis
|
||||||
|
|
||||||
|
- User-provided remote log showed repeated `$'\r': command not found` in `patch_ops.sh`.
|
||||||
|
- Root cause: the shell script, and likely copied `.py` patch helpers, arrived on Linux with Windows CRLF line endings.
|
||||||
|
- Consequence: `python3 ./patch_model_runner.py\r` and similar patch commands did not execute correctly.
|
||||||
|
- Later server startup failed with `unrecognized arguments: --reasoning-parser qwen3`.
|
||||||
|
- Interpretation: OpenAI API runtime files such as `cli_args.py` and `api_server.py` were not patched into `/usr/local/corex/lib/python3/dist-packages/vllm/...`.
|
||||||
|
- Next action: normalize CRLF to LF on the remote copy, rerun `patch_ops.sh`, verify `--reasoning-parser` exists in the installed vLLM CLI, then restart the API server.
|
||||||
|
|
||||||
|
## 2026-07-14 First Server Start After CRLF Fix
|
||||||
|
|
||||||
|
- User verified `python3 -m vllm.entrypoints.openai.api_server --help | grep reasoning-parser` now prints `--reasoning-parser`.
|
||||||
|
- Startup log shows `reasoning_parser='qwen3'`, so OpenAI API argument patch is active.
|
||||||
|
- `ixsmi` shows no vLLM GPU process and `/health` returns `Connection refused`, so the API server process exited before serving.
|
||||||
|
- Need inspect the tail of `/root/work/logs/server_first_run.log` after `Downcasting torch.float32 to torch.float16`.
|
||||||
|
- Additional suspicion to verify: some patches target `/usr/local/corex/lib/python3/dist-packages` while the Qwen3_5 registry patch targets `/usr/local/corex/lib64/python3/dist-packages`; confirm the actual imported registry path and whether both `lib` and `lib64` contain `Qwen3_5MoeForCausalLM`.
|
||||||
|
|
||||||
|
## 2026-07-14 Server Still Initializing
|
||||||
|
|
||||||
|
- User checked `jobs -l` and `ps`; master process PID 2052 is still running.
|
||||||
|
- Log has no `Traceback`, `ERROR`, or common fatal exceptions.
|
||||||
|
- Log reached `Worker ready; awaiting tasks` for three worker processes, so multiprocess executor startup is progressing.
|
||||||
|
- Both `/usr/local/corex/lib/.../vllm` and `/usr/local/corex/lib64/.../vllm` contain `registry.py` patched with `Qwen3_5MoeForCausalLM`, and `qwen3_5.py` exists in both.
|
||||||
|
- Earlier `/health` failure was likely checked before API server finished engine initialization and started listening.
|
||||||
|
- Next action: wait and monitor model loading/GPU memory, then retry `/health`; only diagnose hang if no new log/GPU memory movement for several minutes.
|
||||||
|
|
||||||
|
## 2026-07-14 Port Occupied By Stale First Process
|
||||||
|
|
||||||
|
- User started a second server while PID 2052 from the first run was still alive.
|
||||||
|
- Second run exited with `OSError: [Errno 98] Address already in use` at `sock.bind(("", args.port))`.
|
||||||
|
- `ps` still shows PID 2052 holding the API server command, but `ixsmi` shows no model GPU memory/processes.
|
||||||
|
- Interpretation: PID 2052 is a stale or stuck master process occupying port 1111, not a healthy loaded model service.
|
||||||
|
- Next action: stop PID 2052 and any VllmWorkerProcess children, verify port 1111 is free, restart once with a fresh log filename, then wait for explicit `Uvicorn running` / startup-complete logs before testing `/health`.
|
||||||
|
|
||||||
|
## 2026-07-14 New Instance Startup Fix
|
||||||
|
|
||||||
|
- New host: `ssh-8d2ae743.default.gpu.phanthy.com`.
|
||||||
|
- Initial failure: `api_server.py: error: unrecognized arguments: --reasoning-parser qwen3`.
|
||||||
|
- Cause 1: new instance runtime had not yet applied `qwen3_6_scripts/patch_ops.sh`.
|
||||||
|
- Applied `patch_ops.sh`; `cli_args.py` in both `/usr/local/corex/lib/...` and `/usr/local/corex/lib64/...` then contained `--reasoning-parser`.
|
||||||
|
- Cause 2: service was started from repository root `/root/data-disk-1/enginex-vllm-bi100-qwen36`, whose local `./vllm` package shadowed the patched CoreX runtime vLLM.
|
||||||
|
- Fix: start service from `/root` and set CoreX `PYTHONPATH`/`LD_LIBRARY_PATH`, using `python3 -B` to avoid stale bytecode.
|
||||||
|
- Working log: `/root/work/logs/server_from_root_B.log`.
|
||||||
|
- Health check reached `GET /health HTTP/1.1" 200 OK`.
|
||||||
|
- Minimal `/v1/chat/completions` request reached `POST /v1/chat/completions HTTP/1.1" 200 OK`.
|
||||||
|
- `ixsmi` shows model loaded at roughly 29GB per BI-V100 card.
|
||||||
|
|
||||||
|
## 2026-07-14 Baseline Smoke And Mini Benchmark
|
||||||
|
|
||||||
|
- Remote benchmark script copied to `/root/work/remote_smoke_bench.py`.
|
||||||
|
- Result file: `/root/work/logs/baseline_smoke_20260713_195201.json`.
|
||||||
|
- `/health` status: `200`.
|
||||||
|
- Non-stream smoke request: `200`, 16 completion tokens in `2.41s`, about `6.63 tok/s`.
|
||||||
|
- Streaming 3-run summary:
|
||||||
|
- TTFT average: `1.03s`.
|
||||||
|
- TTFT P90 from 3 samples: about `0.80s` using the simple small-sample estimator.
|
||||||
|
- Output TPS after TTFT: about `8.91 tok/s`.
|
||||||
|
- Prefix-cache probe:
|
||||||
|
- First long-prefix request: `cached_tokens=0`, 16 completion tokens in `6.59s`.
|
||||||
|
- Second related long-prefix request: `cached_tokens=1056 / 1069 prompt tokens`, 16 completion tokens in `3.61s`.
|
||||||
|
- Server metrics also reported GPU prefix cache hit rate around `50.36%`.
|
||||||
|
- Current status: model is fully runnable and API-compatible enough for smoke tests, but generation throughput is far below the contest target of Output TPS P10 >= 20.
|
||||||
|
- First performance direction: reduce thinking-token waste, profile decode path, verify XFormers fallback cost, then tune serving parameters after a larger benchmark.
|
||||||
|
|
||||||
|
## 2026-07-14 Official-Like Benchmark Dataset Preparation
|
||||||
|
|
||||||
|
- Current repository dataset `chat_dataset_v0.json` is only a 4-conversation smoke dataset.
|
||||||
|
- It has short multi-turn chat/role-play prompts and no realistic `tools`, assistant `tool_calls`, or `tool` role messages.
|
||||||
|
- It is not representative of the official benchmark, which is an 881-request long-context Agent coding workload.
|
||||||
|
- Added `worklogs/formal_perf_bench.py` support for OpenAI-style request items with `messages`, `tools`, `tool_choice`, `stream`, and per-request `max_tokens`.
|
||||||
|
- Added `worklogs/generate_official_like_dataset.py` to synthesize official-like long-context, stream=true, tool-heavy requests.
|
||||||
|
- Uploaded updated scripts to:
|
||||||
|
- `/root/work/formal_perf_bench.py`
|
||||||
|
- `/root/work/generate_official_like_dataset.py`
|
||||||
|
- Server smoke generation succeeded:
|
||||||
|
- `/root/work/logs/synthetic_agent_perf_16.jsonl`
|
||||||
|
- 16 requests, 13 sessions, approx prompt average `26611.5`, approximate weighted cache ratio `0.638`.
|
||||||
|
- File size `4.5M`.
|
||||||
|
|
||||||
|
## 2026-07-14 Official-Like Synthetic Benchmark Round 1
|
||||||
|
|
||||||
|
- Fixed synthetic request format issues:
|
||||||
|
- Assistant tool-call messages now use `content: ""` instead of `content: null`.
|
||||||
|
- Synthetic system/context content is merged into a single leading system message because the current runtime rejects multiple system messages with `System message must be at the beginning`.
|
||||||
|
- Added `--max-prompt-tokens` to cap generated prompt sizes under the current server `--max-model-len 100000`.
|
||||||
|
- Generated dataset:
|
||||||
|
- `/root/work/logs/synthetic_agent_perf_16_fixed2_cap90k.jsonl`
|
||||||
|
- 16 requests, approximate average prompt tokens `25712`, approximate intended cache ratio `0.637`.
|
||||||
|
- Benchmark:
|
||||||
|
- Command output file: `/root/work/logs/perf_synth16_fixed2_cap90k_c1_r4.json`.
|
||||||
|
- Concurrency `1`, requests `4`, max output tokens `64`.
|
||||||
|
- Results:
|
||||||
|
- Success rate: `100%` (`4/4`).
|
||||||
|
- Prompt tokens total: `166080`.
|
||||||
|
- Completion tokens total: `256`.
|
||||||
|
- Cached tokens total: `96`.
|
||||||
|
- Actual cache hit rate: `0.058%`, far below official-like target; generator does not yet create exact cumulative session-prefix reuse.
|
||||||
|
- TTFT P90: `129.07s`.
|
||||||
|
- Output TPS P10 per request: `5.23 tok/s`.
|
||||||
|
- Aggregate output TPS: `0.607 tok/s`.
|
||||||
|
- Aggregate uncached input TPS: `393.52 tok/s`.
|
||||||
|
- Weighted token throughput: `1111.78`, far below `8000`.
|
||||||
|
- Interpretation:
|
||||||
|
- Long-context uncached prefill is the immediate bottleneck.
|
||||||
|
- Current synthetic generator must be improved to make same-session adjacent requests share exact prefix, otherwise cache behavior is not representative of the official 65.6% cached-token workload.
|
||||||
|
- Current runtime also needs template compatibility work for multiple/irregular system messages if official requests include them unmodified.
|
||||||
|
|
||||||
|
## 2026-07-14 Cumulative Prefix Synthetic Benchmark Round 2
|
||||||
|
|
||||||
|
- Added `--cumulative` mode to `generate_official_like_dataset.py`.
|
||||||
|
- New mode shares exact system/tool prefix and grows each session history in place, so same-session later requests actually reuse earlier token prefixes.
|
||||||
|
- Generated dataset:
|
||||||
|
- `/root/work/logs/synthetic_cumulative_8_s3_cap40k.jsonl`
|
||||||
|
- 8 requests, 3 sessions, approximate prompt average `27452.75`.
|
||||||
|
- Prompt cap `40000` to keep test runtime manageable under current `--max-model-len 100000`.
|
||||||
|
- Benchmark:
|
||||||
|
- Result file: `/root/work/logs/perf_cumulative8_s3_cap40k_c1_r4.json`.
|
||||||
|
- Concurrency `1`, requests `4`, max output tokens `64`.
|
||||||
|
- Results:
|
||||||
|
- Success rate: `100%` (`4/4`).
|
||||||
|
- TTFT P90: `58.49s`.
|
||||||
|
- Output TPS P10 per request: `5.94 tok/s`.
|
||||||
|
- Aggregate output TPS: `1.55 tok/s`.
|
||||||
|
- Aggregate uncached input TPS: `284.96 tok/s`.
|
||||||
|
- Aggregate cache TPS: `296.87 tok/s`.
|
||||||
|
- Cached tokens: `49152 / 96333`, cache hit rate `51.02%`.
|
||||||
|
- Weighted token throughput: `989.82`.
|
||||||
|
- Target comparison:
|
||||||
|
- Cache hit rate and success rate pass.
|
||||||
|
- TTFT, Output TPS P10, and weighted throughput are far below target.
|
||||||
|
- Interpretation:
|
||||||
|
- Prefix cache is now being exercised correctly, but cached prefill is still too slow at this configuration.
|
||||||
|
- Current serving parameter `--max-num-seqs 1` prevents useful concurrent batching and likely caps weighted throughput.
|
||||||
|
- Decode path is also slow: request-level output TPS only `5.9-8.5 tok/s`.
|
||||||
|
- Next experiment should restart service with higher `--max-num-seqs`, higher `--max-num-batched-tokens`, and possibly `gpu-memory-utilization 0.95`, then rerun the same cumulative dataset.
|
||||||
|
|
||||||
|
## 2026-07-14 Serving Parameter Experiment: seq2 / batched8192
|
||||||
|
|
||||||
|
- Restarted service from `/root` with:
|
||||||
|
- `--gpu-memory-utilization 0.95`
|
||||||
|
- `--max-num-seqs 2`
|
||||||
|
- `--max-num-batched-tokens 8192`
|
||||||
|
- same model, tensor parallel `4`, prefix cache, chunked prefill, and qwen3 parsers.
|
||||||
|
- Service started successfully:
|
||||||
|
- Log file: `/root/work/logs/server_exp_seq2_b8192.log`
|
||||||
|
- Process id observed: `2710`
|
||||||
|
- GPU blocks: `21100`, CPU blocks: `6553`
|
||||||
|
|
||||||
|
### Concurrency 2 result
|
||||||
|
|
||||||
|
- Benchmark:
|
||||||
|
- Dataset: `/root/work/logs/synthetic_cumulative_8_s3_cap40k.jsonl`
|
||||||
|
- Result file: `/root/work/logs/perf_cumulative8_s3_cap40k_exp_seq2_c2_r4.json`
|
||||||
|
- Concurrency `2`, requests `4`, max output tokens `64`.
|
||||||
|
- Results:
|
||||||
|
- Success rate: `100%`.
|
||||||
|
- TTFT P90: `58.51s`.
|
||||||
|
- Output TPS P10 per request: `0.87 tok/s`.
|
||||||
|
- Aggregate output TPS: `1.53 tok/s`.
|
||||||
|
- Cache hit rate: `43.22%`.
|
||||||
|
- Weighted token throughput: `1076.71`.
|
||||||
|
- Interpretation:
|
||||||
|
- Raising concurrency to `2` harms per-request decode speed and does not improve TTFT on this small long-context test.
|
||||||
|
- Cache hit rate also drops below target because overlapping requests disturb the ideal same-session sequential cache path.
|
||||||
|
|
||||||
|
### Concurrency 1 result
|
||||||
|
|
||||||
|
- Benchmark:
|
||||||
|
- Dataset: `/root/work/logs/synthetic_cumulative_8_s3_cap40k.jsonl`
|
||||||
|
- Result file: `/root/work/logs/perf_cumulative8_s3_cap40k_exp_seq2_c1_r4.json`
|
||||||
|
- Concurrency `1`, requests `4`, max output tokens `64`.
|
||||||
|
- Results:
|
||||||
|
- Success rate: `100%`.
|
||||||
|
- TTFT P90: `3.34s`.
|
||||||
|
- Output TPS P10 per request: `6.10 tok/s`.
|
||||||
|
- Aggregate output TPS: `5.50 tok/s`.
|
||||||
|
- Cache hit rate: `99.97%`.
|
||||||
|
- Cached tokens: `96304 / 96333`.
|
||||||
|
- Weighted token throughput: `1252.05`.
|
||||||
|
- Interpretation:
|
||||||
|
- With warm/exact prefix cache reuse, TTFT and cache-hit targets can pass.
|
||||||
|
- Output TPS is still far below the `>=20` target, so the next optimization area is decode throughput.
|
||||||
|
- Weighted throughput remains far below `8000`; this requires either much higher input/cache throughput under realistic scheduling or substantially faster decode.
|
||||||
|
|
||||||
|
## 2026-07-14 Failed Decode Scheduler Experiment: multi-step
|
||||||
|
|
||||||
|
- Investigated available CoreX/vLLM parameters from `/root/work/logs/api_server_help.txt`.
|
||||||
|
- Tried adding:
|
||||||
|
- `--num-scheduler-steps 4`
|
||||||
|
- existing `--enable-chunked-prefill`
|
||||||
|
- Result:
|
||||||
|
- Service failed during startup.
|
||||||
|
- Log file: `/root/work/logs/server_exp_sched4_seq2_b8192.log`
|
||||||
|
- Error: `Multi-Step + Chunked-Prefill not supported for attention backend: xformers`.
|
||||||
|
- The runtime suggests `flash-attn`, but current service logs show `Using XFormers backend`.
|
||||||
|
- Recovery:
|
||||||
|
- Restarted the previously working `seq2/b8192` service without `--num-scheduler-steps`.
|
||||||
|
- Recovery log file: `/root/work/logs/server_exp_seq2_b8192_recover.log`
|
||||||
|
- `/health` returned `200`.
|
||||||
|
- Server log recorded a short `/v1/chat/completions` request returning `200 OK`.
|
||||||
|
- Interpretation:
|
||||||
|
- Multi-step scheduling is not available with the current xFormers + chunked-prefill path.
|
||||||
|
- Next decode optimization should focus on either enabling a compatible attention backend, checking whether chunked prefill can be disabled for a decode-focused variant, or optimizing parser/reasoning/tool overhead before deeper code changes.
|
||||||
|
|
||||||
|
## 2026-07-14 Decode Microbenchmark Round
|
||||||
|
|
||||||
|
- Added local and remote script:
|
||||||
|
- Local: `worklogs/decode_microbench.py`
|
||||||
|
- Remote: `/root/work/decode_microbench.py`
|
||||||
|
- Synced raw result JSON files to:
|
||||||
|
- `worklogs/remote_results/2026-07-14-decode/`
|
||||||
|
- Wrote Chinese analysis report:
|
||||||
|
- `worklogs/decode_analysis_report_2026-07-14.md`
|
||||||
|
- Key results:
|
||||||
|
- Full parser, short prompt, concurrency 1, 256 output tokens: Output TPS P10 `8.74 tok/s`, TTFT P90 `1.85s`.
|
||||||
|
- Full parser with 29 tools, concurrency 1, 128 output tokens: Output TPS P10 `8.70 tok/s`, TTFT P90 `4.98s`.
|
||||||
|
- Parser-off, short prompt, concurrency 1, 256 output tokens: Output TPS P10 `8.30 tok/s`, TTFT P90 `3.18s`.
|
||||||
|
- Parser-off concurrency curve with 128 output tokens:
|
||||||
|
- c1: aggregate Output TPS `8.00`, per-request P10 `8.30`.
|
||||||
|
- c2: aggregate Output TPS `7.23`, per-request P10 `3.68`.
|
||||||
|
- c4: aggregate Output TPS `5.08`, per-request P10 `2.57`, TTFT P90 `51.60s`.
|
||||||
|
- c2 with `ixsmi` sampling: average GPU utilization `28.91%`, max GPU utilization `100%`, average power `49.67W`.
|
||||||
|
- Interpretation:
|
||||||
|
- Tool/reasoning parser is not the decode throughput bottleneck.
|
||||||
|
- Increasing concurrency does not improve aggregate decode throughput and worsens per-request TPS.
|
||||||
|
- Low average GPU utilization suggests scheduling, TP synchronization, MoE kernel, paged attention, or xFormers backend limitations before raw compute saturation.
|
||||||
|
- Next focused experiment: disable chunked prefill and retry `--num-scheduler-steps 4` as a decode-only diagnostic variant.
|
||||||
|
|
||||||
|
## 2026-07-14 Multi-Step Scheduling Diagnostic
|
||||||
|
|
||||||
|
- Goal: verify whether `--num-scheduler-steps 4` can improve decode throughput.
|
||||||
|
- Appended details to:
|
||||||
|
- `worklogs/decode_analysis_report_2026-07-14.md`
|
||||||
|
- Synced remote failure logs to:
|
||||||
|
- `worklogs/remote_results/2026-07-14-scheduler/`
|
||||||
|
- Experiments:
|
||||||
|
- No chunked prefill + `max_model_len=100000` failed because `max_num_batched_tokens=8192` is smaller than `max_model_len=100000`.
|
||||||
|
- No chunked prefill + `max_model_len=4096` still failed because `Multi-Step not supported for attention backend: xformers`.
|
||||||
|
- Forced `VLLM_ATTENTION_BACKEND=FLASHINFER` failed because `BatchDecodeWithPagedKVCacheWrapper` was `None`, indicating FlashInfer dependencies/wrappers are unavailable or incompatible.
|
||||||
|
- Conclusion:
|
||||||
|
- Multi-step scheduling cannot currently be tested or used on this server because the available attention backend is xFormers.
|
||||||
|
- The next practical direction is not more `num_scheduler_steps` tuning, but either making FlashAttention/FlashInfer backend available or profiling the current xFormers + MoE + TP decode path.
|
||||||
|
- Recovery:
|
||||||
|
- Restored full parser service with chunked prefill and prefix cache.
|
||||||
|
- Recovery log: `/root/work/logs/server_exp_seq2_b8192_fullparser_restored_after_schedtest.log`.
|
||||||
|
- `/health` returned `200`.
|
||||||
733
worklogs/decode_analysis_report_2026-07-14.md
Normal file
733
worklogs/decode_analysis_report_2026-07-14.md
Normal file
@@ -0,0 +1,733 @@
|
|||||||
|
# Decode 吞吐专项实验分析报告
|
||||||
|
|
||||||
|
日期:2026-07-14
|
||||||
|
|
||||||
|
## 一、结论摘要
|
||||||
|
|
||||||
|
本轮实验确认:当前服务的主要短板是 decode 阶段吞吐,而不是 prompt prefill、tool parser 或 reasoning parser。
|
||||||
|
|
||||||
|
核心证据:
|
||||||
|
|
||||||
|
- 短 prompt、单并发、256 token 输出时,Output TPS P10 只有 `8.3-8.7 tok/s`,明显低于官方目标 `>=20 tok/s`。
|
||||||
|
- 去掉 `--enable-auto-tool-choice`、`--tool-call-parser`、`--reasoning-parser` 后,decode TPS 没有提升,反而略低。
|
||||||
|
- 加并发后没有获得 batch 增益:并发 1 聚合约 `8.00 tok/s`,并发 2 聚合约 `7.23 tok/s`,并发 4 聚合约 `5.08 tok/s`。
|
||||||
|
- 并发 2 时硬件采样显示平均 GPU-Util 只有 `28.9%`,平均功耗约 `49.7W / 250W`,说明 GPU 没有被持续打满。
|
||||||
|
|
||||||
|
初步判断:瓶颈更像是 decode 路径中的调度、TP 同步、MoE kernel/专家路由、paged attention/kernel launch 开销,或 xFormers 后端限制,而不是 API parser 层。
|
||||||
|
|
||||||
|
## 二、实验环境
|
||||||
|
|
||||||
|
远端服务:
|
||||||
|
|
||||||
|
- 模型:`/root/public-storage/models/Qwen/Qwen3.6-35B-A3B`
|
||||||
|
- API:`vllm.entrypoints.openai.api_server`
|
||||||
|
- GPU:4 x Iluvatar BI-V100, 32GB
|
||||||
|
- 当前主要实验配置:
|
||||||
|
- `-tp 4`
|
||||||
|
- `--gpu-memory-utilization 0.95`
|
||||||
|
- `--max-num-seqs 2`
|
||||||
|
- `--max-num-batched-tokens 8192`
|
||||||
|
- `--enable-chunked-prefill`
|
||||||
|
- `--enable-prefix-caching`
|
||||||
|
|
||||||
|
本地脚本:
|
||||||
|
|
||||||
|
- `worklogs/decode_microbench.py`
|
||||||
|
|
||||||
|
远端脚本:
|
||||||
|
|
||||||
|
- `/root/work/decode_microbench.py`
|
||||||
|
|
||||||
|
本地原始结果:
|
||||||
|
|
||||||
|
- `worklogs/remote_results/2026-07-14-decode/`
|
||||||
|
|
||||||
|
## 三、实验结果
|
||||||
|
|
||||||
|
### 1. 纯短 prompt decode 基线
|
||||||
|
|
||||||
|
完整 parser 配置,短 prompt,单并发,3 请求,每请求 `max_tokens=256`。
|
||||||
|
|
||||||
|
结果文件:
|
||||||
|
|
||||||
|
- 远端:`/root/work/logs/decode_full_parser_short_c1_t256_r3.json`
|
||||||
|
- 本地:`worklogs/remote_results/2026-07-14-decode/decode_full_parser_short_c1_t256_r3.json`
|
||||||
|
|
||||||
|
指标:
|
||||||
|
|
||||||
|
| 指标 | 数值 |
|
||||||
|
|---|---:|
|
||||||
|
| 成功率 | 100% |
|
||||||
|
| TTFT P90 | 1.85s |
|
||||||
|
| Output TPS P10 | 8.74 tok/s |
|
||||||
|
| Output TPS P50 | 8.74 tok/s |
|
||||||
|
| 聚合 Output TPS | 8.34 tok/s |
|
||||||
|
| 总 completion tokens | 768 |
|
||||||
|
| reasoning tokens | 768 |
|
||||||
|
|
||||||
|
解释:
|
||||||
|
|
||||||
|
短 prompt 下 TTFT 已经较低,但 decode 速度仍只有约 `8.7 tok/s`。这说明长上下文不是唯一问题,短输出生成本身就偏慢。
|
||||||
|
|
||||||
|
### 2. Tool/parser 触发场景
|
||||||
|
|
||||||
|
完整 parser 配置,携带 29 个 tools,单并发,3 请求,每请求 `max_tokens=128`。
|
||||||
|
|
||||||
|
结果文件:
|
||||||
|
|
||||||
|
- 远端:`/root/work/logs/decode_full_parser_tool_c1_t128_r3.json`
|
||||||
|
- 本地:`worklogs/remote_results/2026-07-14-decode/decode_full_parser_tool_c1_t128_r3.json`
|
||||||
|
|
||||||
|
指标:
|
||||||
|
|
||||||
|
| 指标 | 数值 |
|
||||||
|
|---|---:|
|
||||||
|
| 成功率 | 100% |
|
||||||
|
| TTFT P90 | 4.98s |
|
||||||
|
| Output TPS P10 | 8.70 tok/s |
|
||||||
|
| Output TPS P50 | 8.70 tok/s |
|
||||||
|
| 聚合 Output TPS | 7.38 tok/s |
|
||||||
|
| prompt tokens | 6639 |
|
||||||
|
| cached tokens | 4416 |
|
||||||
|
| completion tokens | 384 |
|
||||||
|
|
||||||
|
单请求细节:
|
||||||
|
|
||||||
|
- 第 1 个请求:TTFT `5.97s`,cached tokens `0`
|
||||||
|
- 第 2/3 个请求:TTFT 约 `1.0s`,cached tokens `2208`
|
||||||
|
|
||||||
|
解释:
|
||||||
|
|
||||||
|
工具 schema 会明显影响首次 prefill/TTFT,但前缀缓存命中后 TTFT 恢复。Output TPS 仍约 `8.7 tok/s`,与纯短 prompt 基本一致,所以 tool parser 不是 decode TPS 主瓶颈。
|
||||||
|
|
||||||
|
### 3. Parser-off 对照
|
||||||
|
|
||||||
|
重启服务,去掉:
|
||||||
|
|
||||||
|
- `--enable-auto-tool-choice`
|
||||||
|
- `--tool-call-parser qwen3_coder`
|
||||||
|
- `--reasoning-parser qwen3`
|
||||||
|
|
||||||
|
其余参数保持一致。
|
||||||
|
|
||||||
|
短 prompt,单并发,3 请求,每请求 `max_tokens=256`。
|
||||||
|
|
||||||
|
结果文件:
|
||||||
|
|
||||||
|
- 远端:`/root/work/logs/decode_noparser_short_c1_t256_r3.json`
|
||||||
|
- 本地:`worklogs/remote_results/2026-07-14-decode/decode_noparser_short_c1_t256_r3.json`
|
||||||
|
|
||||||
|
指标:
|
||||||
|
|
||||||
|
| 指标 | 完整 parser | parser-off |
|
||||||
|
|---|---:|---:|
|
||||||
|
| 成功率 | 100% | 100% |
|
||||||
|
| TTFT P90 | 1.85s | 3.18s |
|
||||||
|
| Output TPS P10 | 8.74 | 8.30 |
|
||||||
|
| Output TPS P50 | 8.74 | 8.33 |
|
||||||
|
| 聚合 Output TPS | 8.34 | 7.87 |
|
||||||
|
|
||||||
|
解释:
|
||||||
|
|
||||||
|
关闭 parser 没有改善 decode。parser/reasoning/tool 相关启动项不是当前 decode 吞吐低的主因。
|
||||||
|
|
||||||
|
### 4. 并发曲线
|
||||||
|
|
||||||
|
parser-off 服务,短 prompt,每请求 `max_tokens=128`。
|
||||||
|
|
||||||
|
结果文件:
|
||||||
|
|
||||||
|
- `decode_noparser_short_c1_t128_r4.json`
|
||||||
|
- `decode_noparser_short_c2_t128_r4.json`
|
||||||
|
- `decode_noparser_short_c4_t128_r4.json`
|
||||||
|
|
||||||
|
指标:
|
||||||
|
|
||||||
|
| 并发 | 请求数 | 成功率 | TTFT P90 | Output TPS P10/请求 | 聚合 Output TPS |
|
||||||
|
|---:|---:|---:|---:|---:|---:|
|
||||||
|
| 1 | 4 | 100% | 0.79s | 8.30 | 8.00 |
|
||||||
|
| 2 | 4 | 100% | 0.92s | 3.68 | 7.23 |
|
||||||
|
| 4 | 4 | 100% | 51.60s | 2.57 | 5.08 |
|
||||||
|
|
||||||
|
解释:
|
||||||
|
|
||||||
|
并发提高后没有形成有效 batch 增益。并发 2 时单请求 TPS 近似减半,聚合 TPS 也略降;并发 4 时出现明显排队,TTFT P90 被拉到 `51.6s`。
|
||||||
|
|
||||||
|
这说明当前配置下 decode 并发能力很弱。由于服务参数 `--max-num-seqs=2`,并发 4 的后两个请求排队符合预期;但并发 2 聚合吞吐仍不提升,说明 decode 内部没有把双请求 batch 变成更高硬件利用率。
|
||||||
|
|
||||||
|
### 5. 硬件利用率采样
|
||||||
|
|
||||||
|
parser-off 服务,并发 2,4 请求,每请求 `max_tokens=128`,同时采样 `ixsmi`。
|
||||||
|
|
||||||
|
结果文件:
|
||||||
|
|
||||||
|
- 远端:`/root/work/logs/decode_noparser_short_c2_t128_r4_monitor.json`
|
||||||
|
- 远端:`/root/work/logs/ixsmi_noparser_short_c2_t128_r4_monitor.json`
|
||||||
|
- 本地同名文件位于:`worklogs/remote_results/2026-07-14-decode/`
|
||||||
|
|
||||||
|
指标:
|
||||||
|
|
||||||
|
| 指标 | 数值 |
|
||||||
|
|---|---:|
|
||||||
|
| 成功率 | 100% |
|
||||||
|
| TTFT P90 | 0.92s |
|
||||||
|
| Output TPS P10 | 3.79 tok/s |
|
||||||
|
| 聚合 Output TPS | 7.40 tok/s |
|
||||||
|
| ixsmi records | 52 |
|
||||||
|
| parsed GPU samples | 208 |
|
||||||
|
| 平均 GPU-Util | 28.91% |
|
||||||
|
| 最大 GPU-Util | 100% |
|
||||||
|
| 平均显存 | 30477 MiB |
|
||||||
|
| 最大显存 | 30555 MiB |
|
||||||
|
| 平均功耗 | 49.67 W |
|
||||||
|
| 最大功耗 | 51 W |
|
||||||
|
|
||||||
|
解释:
|
||||||
|
|
||||||
|
GPU 利用率和功耗都偏低。虽然瞬时 GPU-Util 能到 100%,但平均只有约 29%,功耗长期接近空载到轻载水平。这说明 decode 过程中存在大量空泡、同步等待或小 kernel 启动开销,GPU 算力没有被持续喂满。
|
||||||
|
|
||||||
|
## 四、瓶颈判断
|
||||||
|
|
||||||
|
当前最可能的瓶颈排序:
|
||||||
|
|
||||||
|
1. Decode 调度/后端限制
|
||||||
|
- `--num-scheduler-steps 4` 曾尝试失败。
|
||||||
|
- 报错:`Multi-Step + Chunked-Prefill not supported for attention backend: xformers`。
|
||||||
|
- 当前服务日志显示使用 `XFormers backend`。
|
||||||
|
|
||||||
|
2. TP 通信或同步开销
|
||||||
|
- 模型使用 `-tp 4`。
|
||||||
|
- 短 decode 每步都可能涉及多卡同步。
|
||||||
|
- 并发 2 聚合 TPS 不升反降,符合小 batch 多卡同步效率差的特征。
|
||||||
|
|
||||||
|
3. MoE decode kernel/专家路由效率
|
||||||
|
- Qwen3.6-35B-A3B 是 MoE 模型。
|
||||||
|
- decode batch 小时,专家路由和 fused MoE kernel 可能难以形成高利用率。
|
||||||
|
|
||||||
|
4. Paged attention / attention backend 每 token 开销
|
||||||
|
- 当前 attention 后端为 xFormers。
|
||||||
|
- multi-step decode 被 xFormers + chunked prefill 组合限制。
|
||||||
|
|
||||||
|
5. API/parser 层
|
||||||
|
- 本轮实验基本排除其为主瓶颈。
|
||||||
|
- tool/schema 影响首次 TTFT,但不显著影响 decode TPS。
|
||||||
|
|
||||||
|
## 五、下一步建议
|
||||||
|
|
||||||
|
建议下一步不要直接改大段 kernel,而是先做两个能明确指向代码修改方向的服务变体实验。
|
||||||
|
|
||||||
|
### 实验 A:decode-only multi-step 变体
|
||||||
|
|
||||||
|
目的:验证 multi-step scheduling 是否能明显提升 decode。
|
||||||
|
|
||||||
|
做法:
|
||||||
|
|
||||||
|
- 暂时关闭 `--enable-chunked-prefill`
|
||||||
|
- 加回 `--num-scheduler-steps 4`
|
||||||
|
- 保持 `max_num_seqs=2`
|
||||||
|
- 跑短 prompt decode 并发 1/2 曲线
|
||||||
|
|
||||||
|
判断:
|
||||||
|
|
||||||
|
- 如果 Output TPS 明显提升,说明 decode 调度是关键方向。
|
||||||
|
- 后续要研究如何让 multi-step 与长上下文 chunked prefill 共存,或按场景切换。
|
||||||
|
|
||||||
|
风险:
|
||||||
|
|
||||||
|
- 官方长上下文负载仍需要 chunked prefill,所以这不是最终配置,只是定位实验。
|
||||||
|
|
||||||
|
### 实验 B:attention backend 变体
|
||||||
|
|
||||||
|
目的:确认是否可以启用兼容 multi-step 的 attention backend。
|
||||||
|
|
||||||
|
做法:
|
||||||
|
|
||||||
|
- 尝试设置 `VLLM_ATTENTION_BACKEND=FLASH_ATTN`
|
||||||
|
- 或检查 CoreX/xFormers flash attention 能否被 vLLM selector 选中
|
||||||
|
- 如果能启动,再测试 multi-step + chunked prefill
|
||||||
|
|
||||||
|
判断:
|
||||||
|
|
||||||
|
- 如果 flash attention 能启动且 multi-step 可用,优先走 backend 配置/适配路线。
|
||||||
|
- 如果不能启动,需要看 `vllm/attention/selector.py`、`vllm/attention/backends/*` 和 CoreX xFormers 补丁。
|
||||||
|
|
||||||
|
### 实验 C:代码级 profiling
|
||||||
|
|
||||||
|
目的:把低 GPU 利用率归因到具体模块。
|
||||||
|
|
||||||
|
建议插桩位置:
|
||||||
|
|
||||||
|
- `vllm/worker/model_runner.py`
|
||||||
|
- `vllm/worker/multi_step_model_runner.py`
|
||||||
|
- `vllm/model_executor/models/qwen3_moe.py`
|
||||||
|
- `vllm/model_executor/layers/fused_moe/*`
|
||||||
|
- `attention.py`
|
||||||
|
- `paged_attn.py`
|
||||||
|
|
||||||
|
记录每步:
|
||||||
|
|
||||||
|
- model forward 耗时
|
||||||
|
- attention 耗时
|
||||||
|
- MoE/MLP 耗时
|
||||||
|
- sampler 耗时
|
||||||
|
- 每步前后同步耗时
|
||||||
|
|
||||||
|
### 实验 D:服务参数小网格
|
||||||
|
|
||||||
|
目的:确认当前 `max_num_seqs=2` 是否已经是最优。
|
||||||
|
|
||||||
|
建议组合:
|
||||||
|
|
||||||
|
| 参数 | 候选 |
|
||||||
|
|---|---|
|
||||||
|
| `max_num_seqs` | 1, 2, 4 |
|
||||||
|
| `max_num_batched_tokens` | 4096, 8192 |
|
||||||
|
| `chunked_prefill` | on/off |
|
||||||
|
| `num_scheduler_steps` | 1, 4 |
|
||||||
|
|
||||||
|
优先只在短 prompt decode 上跑,快速筛掉无效组合。
|
||||||
|
|
||||||
|
## 六、当前推荐行动
|
||||||
|
|
||||||
|
下一步优先跑:
|
||||||
|
|
||||||
|
1. `chunked_prefill=off + num_scheduler_steps=4`
|
||||||
|
2. 若能启动,跑 decode c1/c2/c4 曲线和 ixsmi 采样
|
||||||
|
3. 若 decode TPS 明显提升,再研究如何兼容官方长上下文 prefill
|
||||||
|
4. 若没有提升,进入 qwen3_moe / fused_moe / attention 的代码级 profiling
|
||||||
|
|
||||||
|
本轮最重要的事实是:GPU 平均利用率只有约 `29%`,所以先不要把问题简单归因为“卡算不动”。更像是当前 decode 执行路径没有把四张卡持续喂满。
|
||||||
|
|
||||||
|
## 七、调度方向验证实验:multi-step decode
|
||||||
|
|
||||||
|
追加日期:2026-07-14
|
||||||
|
|
||||||
|
### 目标
|
||||||
|
|
||||||
|
验证 `--num-scheduler-steps 4` 是否能作为 decode 吞吐提升方向。
|
||||||
|
|
||||||
|
由于上一轮实验显示并发升高没有带来 aggregate TPS 增益,且平均 GPU 利用率只有约 `29%`,multi-step scheduling 是最直接的调度侧候选优化:它理论上可以减少每 token 调度往返和 Python/worker 协调开销,让 decode 连续执行多个 step。
|
||||||
|
|
||||||
|
### 实验 1:关闭 chunked prefill,直接启用 multi-step
|
||||||
|
|
||||||
|
启动变体:
|
||||||
|
|
||||||
|
- `--num-scheduler-steps 4`
|
||||||
|
- 不加 `--enable-chunked-prefill`
|
||||||
|
- `--max-model-len 100000`
|
||||||
|
- `--max-num-batched-tokens 8192`
|
||||||
|
- 其它参数沿用完整 parser 服务配置
|
||||||
|
|
||||||
|
结果:
|
||||||
|
|
||||||
|
- 服务未启动。
|
||||||
|
- 远端日志:`/root/work/logs/server_exp_sched4_nochunk_fullparser.log`
|
||||||
|
- 本地归档:`worklogs/remote_results/2026-07-14-scheduler/server_exp_sched4_nochunk_fullparser.log`
|
||||||
|
|
||||||
|
失败原因:
|
||||||
|
|
||||||
|
```text
|
||||||
|
ValueError: max_num_batched_tokens (8192) is smaller than max_model_len (100000).
|
||||||
|
```
|
||||||
|
|
||||||
|
解释:
|
||||||
|
|
||||||
|
关闭 chunked prefill 后,vLLM 要求 `max_num_batched_tokens >= max_model_len`,否则实际最大可处理序列会被 `max_num_batched_tokens` 限制。这个配置不能用于官方长上下文,也无法进入 decode 压测。
|
||||||
|
|
||||||
|
### 实验 2:decode-only 诊断配置
|
||||||
|
|
||||||
|
为绕过实验 1 的限制,临时降低上下文长度,只用于短 prompt decode 诊断。
|
||||||
|
|
||||||
|
启动变体:
|
||||||
|
|
||||||
|
- `--max-model-len 4096`
|
||||||
|
- `--max-seq-len-to-capture 4096`
|
||||||
|
- `--max-num-batched-tokens 8192`
|
||||||
|
- `--num-scheduler-steps 4`
|
||||||
|
- 不加 `--enable-chunked-prefill`
|
||||||
|
- attention backend 仍为自动选择
|
||||||
|
|
||||||
|
结果:
|
||||||
|
|
||||||
|
- 服务未启动。
|
||||||
|
- 远端日志:`/root/work/logs/server_exp_sched4_nochunk_len4096_fullparser.log`
|
||||||
|
- 本地归档:`worklogs/remote_results/2026-07-14-scheduler/server_exp_sched4_nochunk_len4096_fullparser.log`
|
||||||
|
|
||||||
|
关键日志:
|
||||||
|
|
||||||
|
```text
|
||||||
|
Using XFormers backend.
|
||||||
|
ValueError: Multi-Step not supported for attention backend: xformers.
|
||||||
|
Set VLLM_ATTENTION_BACKEND to a value from ['flash-attn', 'rocm-flash-attn', 'flashinfer'].
|
||||||
|
```
|
||||||
|
|
||||||
|
解释:
|
||||||
|
|
||||||
|
这说明当前环境不是“chunked prefill 与 multi-step 的组合不支持”这么简单,而是 `xformers` attention backend 本身不支持 multi-step worker。只要 attention backend 仍然落到 xFormers,multi-step 调度就无法启用。
|
||||||
|
|
||||||
|
### 实验 3:强制 FlashInfer backend
|
||||||
|
|
||||||
|
启动变体:
|
||||||
|
|
||||||
|
- 环境变量:`VLLM_ATTENTION_BACKEND=FLASHINFER`
|
||||||
|
- `--max-model-len 4096`
|
||||||
|
- `--num-scheduler-steps 4`
|
||||||
|
- 不加 `--enable-chunked-prefill`
|
||||||
|
- 其它参数同实验 2
|
||||||
|
|
||||||
|
结果:
|
||||||
|
|
||||||
|
- 服务未启动。
|
||||||
|
- 远端日志:`/root/work/logs/server_exp_sched4_nochunk_len4096_flashinfer.log`
|
||||||
|
- 本地归档:`worklogs/remote_results/2026-07-14-scheduler/server_exp_sched4_nochunk_len4096_flashinfer.log`
|
||||||
|
|
||||||
|
关键日志:
|
||||||
|
|
||||||
|
```text
|
||||||
|
TypeError: 'NoneType' object is not callable
|
||||||
|
...
|
||||||
|
self._decode_wrapper = BatchDecodeWithPagedKVCacheWrapper(...)
|
||||||
|
```
|
||||||
|
|
||||||
|
解释:
|
||||||
|
|
||||||
|
`vllm/attention/backends/flashinfer.py` 会导入:
|
||||||
|
|
||||||
|
- `flashinfer.BatchDecodeWithPagedKVCacheWrapper`
|
||||||
|
- `flashinfer.decode.CUDAGraphBatchDecodeWithPagedKVCacheWrapper`
|
||||||
|
- `flashinfer.prefill.BatchPrefillWithPagedKVCacheWrapper`
|
||||||
|
- `ixformer.contrib.vllm_flash_attn.flash_attn_varlen_func`
|
||||||
|
|
||||||
|
当前环境中至少有关键 FlashInfer wrapper 没导入成功,导致 wrapper 为 `None`,在 profiling 阶段调用时报错。
|
||||||
|
|
||||||
|
### 实验结论
|
||||||
|
|
||||||
|
本轮没有进入 c1/c2/c4 decode 曲线压测,因为 multi-step 服务在启动阶段就失败。
|
||||||
|
|
||||||
|
结论不是“调度一定无效”,而是:
|
||||||
|
|
||||||
|
1. 当前 xFormers backend 下,multi-step 调度不可用。
|
||||||
|
2. 当前 FlashInfer backend 依赖不完整或与 CoreX 环境不兼容,不能直接替代 xFormers。
|
||||||
|
3. 自动 FlashAttention 也不可用;此前日志已显示 `vllm_flash_attn` 包缺失,因此自动回落到 xFormers。
|
||||||
|
|
||||||
|
因此,调度优化如果要继续推进,前置任务是 attention backend 适配:
|
||||||
|
|
||||||
|
- 路线 A:补齐/修复 CoreX 环境里的 FlashAttention 或 FlashInfer backend。
|
||||||
|
- 路线 B:改造 xFormers backend 或 MultiStepModelRunner,使其支持当前 xFormers 路径。
|
||||||
|
- 路线 C:绕过 multi-step,直接 profile xFormers decode、paged attention、MoE 与 TP 通信开销。
|
||||||
|
|
||||||
|
### 对下一步方向的影响
|
||||||
|
|
||||||
|
短期内,继续调 `--num-scheduler-steps` 没意义;它被 backend 卡住了。
|
||||||
|
|
||||||
|
下一步建议改为两条线并行:
|
||||||
|
|
||||||
|
1. **backend 可用性线**
|
||||||
|
- 检查 `flashinfer` 和 `ixformer.contrib.vllm_flash_attn` 在服务器上的实际导入错误。
|
||||||
|
- 确认官方镜像/包中是否本应包含 `vllm_flash_attn`。
|
||||||
|
- 如果能补齐依赖,再重跑 multi-step decode 曲线。
|
||||||
|
|
||||||
|
2. **代码 profiling 线**
|
||||||
|
- 直接在当前可用 xFormers 路径插桩。
|
||||||
|
- 重点记录每 token decode 中 attention、MoE、sampler、TP 同步的耗时。
|
||||||
|
- 当前平均 GPU 利用率低,profiling 比继续盲调参数更有价值。
|
||||||
|
|
||||||
|
## 2026-07-14:解除强制 eager / custom all-reduce 禁用验证
|
||||||
|
|
||||||
|
### 修改内容
|
||||||
|
|
||||||
|
本轮先处理 `vllm/engine/arg_utils.py` 中两个会掩盖真实性能路径的硬编码:
|
||||||
|
|
||||||
|
```python
|
||||||
|
enforce_eager=True
|
||||||
|
disable_custom_all_reduce=True
|
||||||
|
```
|
||||||
|
|
||||||
|
改为尊重 CLI / dataclass 参数:
|
||||||
|
|
||||||
|
```python
|
||||||
|
enforce_eager=self.enforce_eager
|
||||||
|
disable_custom_all_reduce=self.disable_custom_all_reduce
|
||||||
|
```
|
||||||
|
|
||||||
|
同时更新 `qwen3_6_scripts/patch_xformers_sdpa_seq.py`,让后续重新执行 patchops 时也会保留该行为。服务器实际运行路径已确认:
|
||||||
|
|
||||||
|
```text
|
||||||
|
PYTHONPATH=/usr/local/corex/lib/python3/dist-packages:/usr/local/corex/lib64/python3/dist-packages
|
||||||
|
```
|
||||||
|
|
||||||
|
因此运行时同时 patch 了:
|
||||||
|
|
||||||
|
- `/usr/local/corex/lib/python3/dist-packages/vllm/engine/arg_utils.py`
|
||||||
|
- `/usr/local/corex/lib64/python3/dist-packages/vllm/engine/arg_utils.py`
|
||||||
|
|
||||||
|
远端运行时备份:
|
||||||
|
|
||||||
|
- `arg_utils.py.bak_20260714_eager_allreduce`
|
||||||
|
|
||||||
|
### 实验 1:CUDA Graph + custom all-reduce 同时开启
|
||||||
|
|
||||||
|
启动命令不再带 `--enforce-eager`,也不带 `--disable-custom-all-reduce`。
|
||||||
|
|
||||||
|
日志确认配置已生效:
|
||||||
|
|
||||||
|
```text
|
||||||
|
disable_custom_all_reduce=False
|
||||||
|
enforce_eager=False
|
||||||
|
use_async_output_proc=True
|
||||||
|
```
|
||||||
|
|
||||||
|
结果:服务未能完成启动,长时间卡在 CUDA Graph capture 阶段。
|
||||||
|
|
||||||
|
关键日志:
|
||||||
|
|
||||||
|
```text
|
||||||
|
Capturing the model for CUDA graphs.
|
||||||
|
[W CUDAGraph.cpp:145] Warning: Waiting for pending NCCL work to finish before starting graph capture.
|
||||||
|
```
|
||||||
|
|
||||||
|
判断:当前 Iluvatar BI-V100 + CoreX + xFormers + TP=4 路径下,CUDA Graph capture 不可直接启用。它没有快速报错,而是卡在 graph capture / NCCL pending work 阶段,风险比普通参数不兼容更高。短期不建议继续沿 CUDA Graph 方向盲试。
|
||||||
|
|
||||||
|
本地归档:
|
||||||
|
|
||||||
|
- `worklogs/remote_results/2026-07-14-eager-allreduce/server_exp_graph_allreduce_seq2_b8192.log`
|
||||||
|
|
||||||
|
### 实验 2:仅开启 custom all-reduce,继续 eager
|
||||||
|
|
||||||
|
启动命令保留 `--enforce-eager`,但不再带 `--disable-custom-all-reduce`。
|
||||||
|
|
||||||
|
日志确认:
|
||||||
|
|
||||||
|
```text
|
||||||
|
enforce_eager=True
|
||||||
|
disable_custom_all_reduce=False
|
||||||
|
```
|
||||||
|
|
||||||
|
服务可以正常启动并通过 `/health`。
|
||||||
|
|
||||||
|
decode microbench 结果:
|
||||||
|
|
||||||
|
| 配置 | 成功率 | TTFT P90 | Output TPS P10/req | Aggregate Output TPS | 对比旧结果 |
|
||||||
|
| --- | ---: | ---: | ---: | ---: | --- |
|
||||||
|
| short c1, 256 tok, custom AR on | 100% | 3.47s | 8.47 | 7.94 | 旧 full-parser c1 为 P10 8.74 / aggregate 8.34,略降 |
|
||||||
|
| short c2, 128 tok, custom AR on | 100% | 1.46s | 4.22 | 8.07 | 旧 parser-off c2 aggregate 7.23,略升但口径不完全相同 |
|
||||||
|
|
||||||
|
本地归档:
|
||||||
|
|
||||||
|
- `worklogs/remote_results/2026-07-14-eager-allreduce/decode_eager_custom_ar_short_c1_t256_r3.json`
|
||||||
|
- `worklogs/remote_results/2026-07-14-eager-allreduce/decode_eager_custom_ar_short_c2_t128_r4.json`
|
||||||
|
- `worklogs/remote_results/2026-07-14-eager-allreduce/server_exp_eager_custom_ar_seq2_b8192_retry.log`
|
||||||
|
|
||||||
|
### 结论
|
||||||
|
|
||||||
|
1. 之前的硬编码确实屏蔽了真实配置,本轮已经解除,并确认修改落在实际运行的 CoreX site-packages 路径里。
|
||||||
|
2. CUDA Graph 当前不兼容或存在严重启动卡死问题,不适合作为短期主优化方向。
|
||||||
|
3. custom all-reduce 可以启动和推理,但收益有限:c2 聚合吞吐有小幅提升,c1 无提升。
|
||||||
|
4. 当前 decode 吞吐仍在 8 tok/s 左右,距离 Output TPS P10 >= 20 仍有明显差距,瓶颈不只是 all-reduce 开关。
|
||||||
|
|
||||||
|
### 下一步 profiling 方向
|
||||||
|
|
||||||
|
优先进入代码级 profiling,而不是继续调 CLI 开关:
|
||||||
|
|
||||||
|
1. 在 `ModelRunner.execute_model` 统计模型 forward、logits、sample 的阶段耗时。
|
||||||
|
2. 在 Qwen MoE 层统计 attention、MoE expert、MoE gate、TP all-reduce 的耗时占比。
|
||||||
|
3. 在 fused MoE 路径统计 topk、expert kernel、activation、sum 的耗时。
|
||||||
|
4. 用 `ENGINEX_PROFILE_DECODE=1` 这类环境变量控制插桩,只在短压测时开启,避免污染正式结果。
|
||||||
|
|
||||||
|
初步判断:custom all-reduce 不是第一大瓶颈;更可能的主战场是 xFormers decode attention、MoE 小 batch kernel、以及 TP 下大量小 kernel / 同步造成的低 GPU 利用率。
|
||||||
|
|
||||||
|
## 2026-07-14:代码级 profiling 与第一轮 MoE 优化
|
||||||
|
|
||||||
|
### profiling 插桩
|
||||||
|
|
||||||
|
新增环境变量控制的 profiling:
|
||||||
|
|
||||||
|
- `ENGINEX_PROFILE_DECODE=1`:开启 profiling。
|
||||||
|
- `ENGINEX_PROFILE_EVERY=N`:每 N 次 model forward 打印一次累计统计。
|
||||||
|
- `ENGINEX_PROFILE_SYNC=1`:每段计时前后 `torch.cuda.synchronize()`,用于定位 GPU 时间。
|
||||||
|
|
||||||
|
插桩位置:
|
||||||
|
|
||||||
|
- `vllm/worker/model_runner.py`
|
||||||
|
- `attn_state.begin_forward`
|
||||||
|
- `model_forward`
|
||||||
|
- `compute_logits`
|
||||||
|
- `sample`
|
||||||
|
- `qwen3_6_scripts/qwen3_5.py`
|
||||||
|
- `full_attention`: qkv projection / norm+rope / paged attention / gate+o_proj
|
||||||
|
- `linear_attention`: GatedDeltaNet 整层
|
||||||
|
- `MoE`: gate / routing topk / routed experts / shared expert / TP all-reduce
|
||||||
|
- `DecoderLayer`: norm / attention / MLP
|
||||||
|
|
||||||
|
注意:profiling 强制同步会显著拖慢请求,因此 profiling 结果只用于定位瓶颈,不作为真实性能分数。
|
||||||
|
|
||||||
|
本地归档:
|
||||||
|
|
||||||
|
- `worklogs/remote_results/2026-07-14-code-profile/server_profile_mode_eager_custom_ar_seq2_b8192.log`
|
||||||
|
- `worklogs/remote_results/2026-07-14-code-profile/profile_mode_eager_custom_ar_short_c1_t24_r1.json`
|
||||||
|
|
||||||
|
### 关键 profiling 结果
|
||||||
|
|
||||||
|
短请求 `c1, max_tokens=24`,修正标签后,rank0 在 step=24 的主要累计耗时:
|
||||||
|
|
||||||
|
| 模块 | 总耗时 | 平均单层/次 | 次数 | 判断 |
|
||||||
|
| --- | ---: | ---: | ---: | --- |
|
||||||
|
| `prefill.moe.routed_prefill_experts` | 5239.90ms | 65.50ms | 80 | 首 token 慢的最大来源 |
|
||||||
|
| `prefill.layer.linear_attention` | 2944.76ms | 49.08ms | 60 | prefill 第二大来源 |
|
||||||
|
| `decode.layer.mlp` | 1456.57ms | 1.66ms | 880 | decode 最大来源 |
|
||||||
|
| `decode.layer.linear_attention` | 942.63ms | 1.43ms | 660 | decode 第二大来源 |
|
||||||
|
| `decode.moe.routed_total` | 785.65ms | 0.89ms | 880 | MLP 中 routed expert 为主 |
|
||||||
|
| `decode.moe.routed_decode_experts` | 540.24ms | 0.61ms | 880 | 单 token MoE expert 计算 |
|
||||||
|
| `decode.layer.full_attention` | 322.06ms | 1.46ms | 220 | full attention 不是第一瓶颈 |
|
||||||
|
| `decode.moe.tp_all_reduce` | 265.92ms | 0.30ms | 880 | 通信有成本,但不是最大项 |
|
||||||
|
|
||||||
|
`ModelRunner` 粗粒度:
|
||||||
|
|
||||||
|
```text
|
||||||
|
decode.model_forward avg ~= 145.6ms
|
||||||
|
decode.compute_logits avg ~= 1.17ms
|
||||||
|
decode.sample avg ~= 1.01ms
|
||||||
|
```
|
||||||
|
|
||||||
|
结论:decode 慢主要发生在模型 forward 内部;logits 和 sampler 不是主瓶颈。
|
||||||
|
|
||||||
|
### 为什么 GPU 算力打不满
|
||||||
|
|
||||||
|
当前路径的 GPU 利用率低,不是因为单个大矩阵乘算不过来,而是因为每 token 被拆成大量小工作:
|
||||||
|
|
||||||
|
1. **MoE 没有真正 fused kernel**
|
||||||
|
- 注释里已经说明 BI-V100 上缺少 `vllm_moe_topk_softmax / vllm_invoke_fused_moe_kernel`。
|
||||||
|
- 当前 routed expert 是纯 PyTorch 实现。
|
||||||
|
- prefill 通用路径按 expert 做 Python 循环,长 prompt 下会产生大量小 GEMM 和 CPU/GPU 同步点。
|
||||||
|
|
||||||
|
2. **decode batch 太小**
|
||||||
|
- `max_num_seqs=2` 时每步只有 1-2 token。
|
||||||
|
- 小 batch 下矩阵乘规模小,kernel launch、Python 调度、TP 同步成本占比很高。
|
||||||
|
|
||||||
|
3. **模型结构有大量 GatedDeltaNet linear_attention 层**
|
||||||
|
- profiling 显示 linear_attention 在 prefill 和 decode 都是大头之一。
|
||||||
|
- 这部分不是标准 paged attention,不能靠换 xFormers attention backend 直接解决。
|
||||||
|
|
||||||
|
4. **TP all-reduce 不是第一瓶颈,但放大了小 kernel 问题**
|
||||||
|
- decode MoE all-reduce 单次约 0.30ms。
|
||||||
|
- 单次看不大,但每层一次、每 token 多次累积,且会让 rank 间等待更明显。
|
||||||
|
|
||||||
|
### 第一轮针对性优化:MoE tiny-batch fast path
|
||||||
|
|
||||||
|
发现:原代码只有 `T == 1` 的 MoE decode fast path。一旦并发 decode `T == 2`,会落入通用 `routed_prefill_experts` 路径:
|
||||||
|
|
||||||
|
```python
|
||||||
|
unique_eids = topk_ids.view(-1).unique().tolist()
|
||||||
|
for eid in unique_eids:
|
||||||
|
...
|
||||||
|
```
|
||||||
|
|
||||||
|
这条路径适合大 prefill,但不适合 `max_num_seqs=2` 的小批量 decode。
|
||||||
|
|
||||||
|
本轮新增 `T <= 2` fast path:
|
||||||
|
|
||||||
|
- 将 `T * top_k` 个选中 expert 展平。
|
||||||
|
- 用两次 batched `torch.bmm` 计算 gate/up 和 down。
|
||||||
|
- 避免 Python per-expert loop。
|
||||||
|
- 保留 `T == 1` 原 fast path 不变。
|
||||||
|
|
||||||
|
### 优化结果
|
||||||
|
|
||||||
|
对比同口径 `short c2, max_tokens=128, requests=4`:
|
||||||
|
|
||||||
|
| 版本 | 成功率 | TTFT P90 | per-request Output TPS P10 | Aggregate Output TPS |
|
||||||
|
| --- | ---: | ---: | ---: | ---: |
|
||||||
|
| 优化前 custom all-reduce on | 100% | 1.46s | 4.22 | 8.07 |
|
||||||
|
| tiny-batch MoE fast path | 100% | 4.42s | 5.95 | 10.60 |
|
||||||
|
|
||||||
|
decode 聚合吞吐提升约 **31%**。TTFT 变差可能来自冷缓存/加载后首次请求抖动,后续需要用多轮 warmup 后再复测。
|
||||||
|
|
||||||
|
本地归档:
|
||||||
|
|
||||||
|
- `worklogs/remote_results/2026-07-14-code-profile/decode_tiny_batch_moe_short_c2_t128_r4.json`
|
||||||
|
- `worklogs/remote_results/2026-07-14-code-profile/server_tiny_batch_moe_eager_custom_ar_seq2_b8192.log`
|
||||||
|
|
||||||
|
### 下一步优化方向
|
||||||
|
|
||||||
|
优先级从高到低:
|
||||||
|
|
||||||
|
1. **MoE prefill 路径**
|
||||||
|
- 当前 `prefill.moe.routed_prefill_experts` 是 TTFT 最大来源。
|
||||||
|
- 需要把 Python per-expert loop 替换成更批量化的 grouped GEMM / batched GEMM。
|
||||||
|
- 官方负载长上下文输入占比极高,提升 prefill 会直接改善 TTFT 和 weighted throughput。
|
||||||
|
|
||||||
|
2. **GatedDeltaNet linear_attention**
|
||||||
|
- decode 和 prefill 都是大头。
|
||||||
|
- 需要进一步拆分 projection、conv/update、state update、out_proj,确认是 recurrent update 还是投影占主。
|
||||||
|
|
||||||
|
3. **更稳的并发策略**
|
||||||
|
- tiny-batch MoE 已证明 `T=2` 能受益。
|
||||||
|
- 后续可测试 `max_num_seqs=3/4`,但受 100K 上下文 KV cache 与 TTFT 影响,需要小心。
|
||||||
|
|
||||||
|
4. **TP all-reduce 合并**
|
||||||
|
- 当前 MoE 每层 routed+shared 后做一次 all-reduce。
|
||||||
|
- 如果后续能将若干小通信或 residual 路径合并,可能进一步改善 decode 抖动,但优先级低于 MoE/linear_attention 计算本体。
|
||||||
|
|
||||||
|
## 2026-07-14:MoE decode fast path v2(tokenwise)
|
||||||
|
|
||||||
|
### 背景
|
||||||
|
|
||||||
|
上一轮 `T <= 2` tiny-batch fast path 使用 batched `torch.bmm`,将 `T * top_k` 个 expert 选择展平后批量计算。它已经将 c2 aggregate Output TPS 从 8.07 提升到 10.60。
|
||||||
|
|
||||||
|
继续分析后发现,对于极小 batch(`T=2`):
|
||||||
|
|
||||||
|
- 第一段 gate/up projection 用 `torch.bmm` 实际是 `T * top_k` 个 `1 x H` 小矩阵乘。
|
||||||
|
- 原 `T == 1` 路径用的是 `F.linear(hidden, w13_sel.reshape(-1, H))`,会形成一个更大的 GEMM,通常更适合 GPU。
|
||||||
|
|
||||||
|
因此新增 v2:`tokenwise` tiny-batch 实现。
|
||||||
|
|
||||||
|
### 实现
|
||||||
|
|
||||||
|
新增环境变量:
|
||||||
|
|
||||||
|
- `ENGINEX_MOE_TINY_IMPL=tokenwise|bmm`
|
||||||
|
- `ENGINEX_MOE_TINY_MAX=4`
|
||||||
|
|
||||||
|
默认:
|
||||||
|
|
||||||
|
```text
|
||||||
|
ENGINEX_MOE_TINY_IMPL=tokenwise
|
||||||
|
ENGINEX_MOE_TINY_MAX=4
|
||||||
|
```
|
||||||
|
|
||||||
|
核心逻辑:
|
||||||
|
|
||||||
|
- 对 `T <= 4` 的小批量 decode,逐 token 走已经验证过的 `T == 1` 大 `F.linear` 路径。
|
||||||
|
- 每个 token 内仍然把 top-k expert 的 `w13` 拼成一个大权重,减少第一段 projection 的小 GEMM。
|
||||||
|
- 保留上一轮 batched bmm 实现,可通过 `ENGINEX_MOE_TINY_IMPL=bmm` 回退做 A/B 测试。
|
||||||
|
|
||||||
|
代码位置:
|
||||||
|
|
||||||
|
- `qwen3_6_scripts/qwen3_5.py`
|
||||||
|
- `Qwen3_5MoeSparseBlock._pure_pytorch_experts`
|
||||||
|
|
||||||
|
### 结果
|
||||||
|
|
||||||
|
同口径 `short c2, max_tokens=128, requests=4`:
|
||||||
|
|
||||||
|
| 版本 | 成功率 | TTFT P90 | per-request Output TPS P10 | Aggregate Output TPS |
|
||||||
|
| --- | ---: | ---: | ---: | ---: |
|
||||||
|
| custom all-reduce on,优化前 | 100% | 1.46s | 4.22 | 8.07 |
|
||||||
|
| tiny-batch bmm | 100% | 4.42s | 5.95 | 10.60 |
|
||||||
|
| tiny-batch tokenwise v2 | 100% | 4.49s | 6.43 | 11.28 |
|
||||||
|
|
||||||
|
相对上一版 bmm,aggregate Output TPS 又提升约 **6.4%**;相对优化前提升约 **39.8%**。
|
||||||
|
|
||||||
|
本地归档:
|
||||||
|
|
||||||
|
- `worklogs/remote_results/2026-07-14-code-profile/decode_moe_tokenwise_short_c2_t128_r4.json`
|
||||||
|
- `worklogs/remote_results/2026-07-14-code-profile/server_moe_tokenwise_eager_custom_ar_seq2_b8192.log`
|
||||||
|
|
||||||
|
### 结论
|
||||||
|
|
||||||
|
这个结果说明:对当前 BI-V100 + PyTorch fallback MoE 路径,**将小 GEMM 尽量合并为较大的 per-token GEMM** 比把所有 token/expert 都塞进 batched bmm 更好。瓶颈确实集中在 decode MoE routed expert 的小矩阵计算与调度开销。
|
||||||
|
|
||||||
|
下一步可继续沿两个方向走:
|
||||||
|
|
||||||
|
1. 测试 `max_num_seqs=4`,利用 `T <= 4` tokenwise fast path,看 aggregate Output TPS 是否继续上涨。
|
||||||
|
2. 继续优化 shared expert / linear_attention,因为 routed expert 已经明显改善,decode 下一个大头会逐渐转向 `linear_attention` 和 shared expert。
|
||||||
314
worklogs/decode_microbench.py
Normal file
314
worklogs/decode_microbench.py
Normal file
@@ -0,0 +1,314 @@
|
|||||||
|
import argparse
|
||||||
|
import concurrent.futures
|
||||||
|
import json
|
||||||
|
import math
|
||||||
|
import os
|
||||||
|
import re
|
||||||
|
import subprocess
|
||||||
|
import threading
|
||||||
|
import time
|
||||||
|
import urllib.error
|
||||||
|
import urllib.request
|
||||||
|
from dataclasses import asdict, dataclass
|
||||||
|
from datetime import datetime
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class RequestResult:
|
||||||
|
ok: bool
|
||||||
|
elapsed_sec: float
|
||||||
|
ttft_sec: float | None
|
||||||
|
completion_tokens: int
|
||||||
|
prompt_tokens: int
|
||||||
|
cached_tokens: int
|
||||||
|
reasoning_tokens: int
|
||||||
|
output_tps: float | None
|
||||||
|
chars: int
|
||||||
|
error: str | None = None
|
||||||
|
|
||||||
|
|
||||||
|
def percentile(values, pct):
|
||||||
|
values = sorted(v for v in values if v is not None)
|
||||||
|
if not values:
|
||||||
|
return None
|
||||||
|
if len(values) == 1:
|
||||||
|
return values[0]
|
||||||
|
pos = (len(values) - 1) * pct / 100.0
|
||||||
|
lo = math.floor(pos)
|
||||||
|
hi = math.ceil(pos)
|
||||||
|
if lo == hi:
|
||||||
|
return values[lo]
|
||||||
|
return values[lo] * (hi - pos) + values[hi] * (pos - lo)
|
||||||
|
|
||||||
|
|
||||||
|
def make_tools(count):
|
||||||
|
tools = []
|
||||||
|
for i in range(count):
|
||||||
|
tools.append({
|
||||||
|
"type": "function",
|
||||||
|
"function": {
|
||||||
|
"name": f"tool_{i:02d}_exec",
|
||||||
|
"description": "Run a deterministic diagnostic action.",
|
||||||
|
"parameters": {
|
||||||
|
"type": "object",
|
||||||
|
"properties": {
|
||||||
|
"command": {"type": "string"},
|
||||||
|
"path": {"type": "string"},
|
||||||
|
},
|
||||||
|
"required": ["command"],
|
||||||
|
},
|
||||||
|
},
|
||||||
|
})
|
||||||
|
return tools
|
||||||
|
|
||||||
|
|
||||||
|
def make_request(args, idx):
|
||||||
|
if args.prompt_mode == "short":
|
||||||
|
user = (
|
||||||
|
"Do not explain. Output a comma-separated sequence of four digit "
|
||||||
|
"numbers starting at 0001. Continue until the token limit stops you."
|
||||||
|
)
|
||||||
|
messages = [{"role": "user", "content": user}]
|
||||||
|
elif args.prompt_mode == "tool":
|
||||||
|
messages = [
|
||||||
|
{"role": "system", "content": "You are a coding agent. Return concise tool-call-like JSON text."},
|
||||||
|
{"role": "user", "content": "Create a shell command to list Python files and print the answer as JSON."},
|
||||||
|
]
|
||||||
|
else:
|
||||||
|
raise ValueError(f"unknown prompt_mode: {args.prompt_mode}")
|
||||||
|
|
||||||
|
payload = {
|
||||||
|
"model": args.model,
|
||||||
|
"messages": messages,
|
||||||
|
"max_tokens": args.max_tokens,
|
||||||
|
"temperature": 0,
|
||||||
|
"stream": True,
|
||||||
|
"stream_options": {"include_usage": True},
|
||||||
|
}
|
||||||
|
if args.with_tools:
|
||||||
|
payload["tools"] = make_tools(args.tool_count)
|
||||||
|
payload["tool_choice"] = "auto"
|
||||||
|
return payload
|
||||||
|
|
||||||
|
|
||||||
|
def post_stream(url, payload, timeout):
|
||||||
|
req = urllib.request.Request(
|
||||||
|
url.rstrip("/") + "/v1/chat/completions",
|
||||||
|
data=json.dumps(payload, ensure_ascii=False).encode("utf-8"),
|
||||||
|
headers={"Content-Type": "application/json"},
|
||||||
|
method="POST",
|
||||||
|
)
|
||||||
|
start = time.perf_counter()
|
||||||
|
first_at = None
|
||||||
|
usage = {}
|
||||||
|
chars = 0
|
||||||
|
try:
|
||||||
|
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
||||||
|
for raw in resp:
|
||||||
|
line = raw.decode("utf-8", errors="replace").strip()
|
||||||
|
if not line or not line.startswith("data:"):
|
||||||
|
continue
|
||||||
|
data = line[5:].strip()
|
||||||
|
if data == "[DONE]":
|
||||||
|
break
|
||||||
|
obj = json.loads(data)
|
||||||
|
if obj.get("usage"):
|
||||||
|
usage = obj["usage"]
|
||||||
|
for choice in obj.get("choices") or []:
|
||||||
|
delta = choice.get("delta") or {}
|
||||||
|
parts = [
|
||||||
|
delta.get("content") or "",
|
||||||
|
delta.get("reasoning_content") or "",
|
||||||
|
]
|
||||||
|
for tool_call in delta.get("tool_calls") or []:
|
||||||
|
fn = tool_call.get("function") or {}
|
||||||
|
parts.append(fn.get("name") or "")
|
||||||
|
parts.append(fn.get("arguments") or "")
|
||||||
|
added = sum(len(p) for p in parts)
|
||||||
|
if added and first_at is None:
|
||||||
|
first_at = time.perf_counter()
|
||||||
|
chars += added
|
||||||
|
elapsed = time.perf_counter() - start
|
||||||
|
ttft = first_at - start if first_at is not None else None
|
||||||
|
details = usage.get("prompt_tokens_details") or {}
|
||||||
|
completion_tokens = int(usage.get("completion_tokens") or 0)
|
||||||
|
decode_sec = elapsed - ttft if ttft is not None else elapsed
|
||||||
|
return RequestResult(
|
||||||
|
ok=True,
|
||||||
|
elapsed_sec=elapsed,
|
||||||
|
ttft_sec=ttft,
|
||||||
|
completion_tokens=completion_tokens,
|
||||||
|
prompt_tokens=int(usage.get("prompt_tokens") or 0),
|
||||||
|
cached_tokens=int(details.get("cached_tokens") or 0),
|
||||||
|
reasoning_tokens=int(usage.get("reasoning_tokens") or 0),
|
||||||
|
output_tps=completion_tokens / decode_sec if completion_tokens and decode_sec > 0 else None,
|
||||||
|
chars=chars,
|
||||||
|
)
|
||||||
|
except urllib.error.HTTPError as exc:
|
||||||
|
body = exc.read().decode("utf-8", errors="replace")[:1000]
|
||||||
|
return RequestResult(False, time.perf_counter() - start, None, 0, 0, 0, 0, None, chars, f"HTTP {exc.code}: {body}")
|
||||||
|
except Exception as exc:
|
||||||
|
return RequestResult(False, time.perf_counter() - start, None, 0, 0, 0, 0, None, chars, f"{type(exc).__name__}: {exc}")
|
||||||
|
|
||||||
|
|
||||||
|
def parse_ixsmi(raw):
|
||||||
|
samples = []
|
||||||
|
for line in raw.splitlines():
|
||||||
|
m = re.search(r"\|\s*\d+%\s+\d+C\s+\S+\s+(\d+)W\s*/\s*(\d+)W\s*\|\s*(\d+)MiB\s*/\s*(\d+)MiB\s*\|\s*(\d+)%", line)
|
||||||
|
if m:
|
||||||
|
samples.append({
|
||||||
|
"power_w": int(m.group(1)),
|
||||||
|
"power_cap_w": int(m.group(2)),
|
||||||
|
"mem_mib": int(m.group(3)),
|
||||||
|
"mem_total_mib": int(m.group(4)),
|
||||||
|
"util_pct": int(m.group(5)),
|
||||||
|
})
|
||||||
|
return samples
|
||||||
|
|
||||||
|
|
||||||
|
def monitor_ixsmi(stop_event, out_path, interval):
|
||||||
|
records = []
|
||||||
|
env = dict(os.environ)
|
||||||
|
env["LD_LIBRARY_PATH"] = ":".join([
|
||||||
|
"/usr/local/corex/lib64",
|
||||||
|
"/usr/local/corex/lib",
|
||||||
|
"/usr/local/iluvatar/lib64",
|
||||||
|
env.get("LD_LIBRARY_PATH", ""),
|
||||||
|
])
|
||||||
|
candidates = [
|
||||||
|
"/usr/local/corex/bin/ixsmi",
|
||||||
|
"/usr/local/iluvatar/bin/ixsmi",
|
||||||
|
"ixsmi",
|
||||||
|
]
|
||||||
|
while not stop_event.is_set():
|
||||||
|
ts = datetime.now().isoformat(timespec="seconds")
|
||||||
|
try:
|
||||||
|
last_error = None
|
||||||
|
raw = None
|
||||||
|
for binary in candidates:
|
||||||
|
try:
|
||||||
|
raw = subprocess.check_output([binary], text=True, stderr=subprocess.STDOUT, timeout=10, env=env)
|
||||||
|
break
|
||||||
|
except Exception as exc:
|
||||||
|
last_error = exc
|
||||||
|
if raw is None:
|
||||||
|
raise last_error or RuntimeError("ixsmi not found")
|
||||||
|
records.append({"ts": ts, "ok": True, "raw": raw, "parsed": parse_ixsmi(raw)})
|
||||||
|
except Exception as exc:
|
||||||
|
records.append({"ts": ts, "ok": False, "error": repr(exc)})
|
||||||
|
stop_event.wait(interval)
|
||||||
|
Path(out_path).write_text(json.dumps(records, ensure_ascii=False, indent=2), encoding="utf-8")
|
||||||
|
|
||||||
|
|
||||||
|
def summarize_monitor(path):
|
||||||
|
if not path or not Path(path).exists():
|
||||||
|
return None
|
||||||
|
records = json.loads(Path(path).read_text(encoding="utf-8"))
|
||||||
|
util = []
|
||||||
|
mem = []
|
||||||
|
power = []
|
||||||
|
for rec in records:
|
||||||
|
for gpu in rec.get("parsed") or []:
|
||||||
|
util.append(gpu["util_pct"])
|
||||||
|
mem.append(gpu["mem_mib"])
|
||||||
|
power.append(gpu["power_w"])
|
||||||
|
if not util:
|
||||||
|
return {"records": len(records), "parsed_samples": 0}
|
||||||
|
return {
|
||||||
|
"records": len(records),
|
||||||
|
"parsed_samples": len(util),
|
||||||
|
"avg_gpu_util_pct": sum(util) / len(util),
|
||||||
|
"max_gpu_util_pct": max(util),
|
||||||
|
"avg_mem_mib": sum(mem) / len(mem),
|
||||||
|
"max_mem_mib": max(mem),
|
||||||
|
"avg_power_w": sum(power) / len(power),
|
||||||
|
"max_power_w": max(power),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
parser = argparse.ArgumentParser()
|
||||||
|
parser.add_argument("--url", default="http://127.0.0.1:1111")
|
||||||
|
parser.add_argument("--model", default="llm")
|
||||||
|
parser.add_argument("--label", required=True)
|
||||||
|
parser.add_argument("--prompt-mode", choices=["short", "tool"], default="short")
|
||||||
|
parser.add_argument("--with-tools", action="store_true")
|
||||||
|
parser.add_argument("--tool-count", type=int, default=16)
|
||||||
|
parser.add_argument("--concurrency", type=int, default=1)
|
||||||
|
parser.add_argument("--requests", type=int, default=4)
|
||||||
|
parser.add_argument("--max-tokens", type=int, default=256)
|
||||||
|
parser.add_argument("--timeout", type=int, default=900)
|
||||||
|
parser.add_argument("--monitor-out")
|
||||||
|
parser.add_argument("--monitor-interval", type=float, default=1.0)
|
||||||
|
parser.add_argument("--out", required=True)
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
stop_event = threading.Event()
|
||||||
|
monitor_thread = None
|
||||||
|
if args.monitor_out:
|
||||||
|
monitor_thread = threading.Thread(
|
||||||
|
target=monitor_ixsmi,
|
||||||
|
args=(stop_event, args.monitor_out, args.monitor_interval),
|
||||||
|
daemon=True,
|
||||||
|
)
|
||||||
|
monitor_thread.start()
|
||||||
|
|
||||||
|
started = time.perf_counter()
|
||||||
|
results = []
|
||||||
|
try:
|
||||||
|
with concurrent.futures.ThreadPoolExecutor(max_workers=args.concurrency) as pool:
|
||||||
|
futures = [
|
||||||
|
pool.submit(post_stream, args.url, make_request(args, i), args.timeout)
|
||||||
|
for i in range(args.requests)
|
||||||
|
]
|
||||||
|
for fut in concurrent.futures.as_completed(futures):
|
||||||
|
result = fut.result()
|
||||||
|
results.append(result)
|
||||||
|
print(f"done {len(results)}/{args.requests} ok={result.ok} tps={result.output_tps}", flush=True)
|
||||||
|
finally:
|
||||||
|
stop_event.set()
|
||||||
|
if monitor_thread:
|
||||||
|
monitor_thread.join(timeout=15)
|
||||||
|
|
||||||
|
wall = time.perf_counter() - started
|
||||||
|
ok = [r for r in results if r.ok]
|
||||||
|
tps_values = [r.output_tps for r in ok if r.output_tps is not None]
|
||||||
|
ttft_values = [r.ttft_sec for r in ok if r.ttft_sec is not None]
|
||||||
|
completion = sum(r.completion_tokens for r in ok)
|
||||||
|
prompt = sum(r.prompt_tokens for r in ok)
|
||||||
|
cached = sum(r.cached_tokens for r in ok)
|
||||||
|
summary = {
|
||||||
|
"created_at": datetime.now().isoformat(timespec="seconds"),
|
||||||
|
"label": args.label,
|
||||||
|
"url": args.url,
|
||||||
|
"model": args.model,
|
||||||
|
"prompt_mode": args.prompt_mode,
|
||||||
|
"with_tools": args.with_tools,
|
||||||
|
"tool_count": args.tool_count if args.with_tools else 0,
|
||||||
|
"concurrency": args.concurrency,
|
||||||
|
"requests": args.requests,
|
||||||
|
"max_tokens": args.max_tokens,
|
||||||
|
"wall_sec": wall,
|
||||||
|
"success_rate": len(ok) / len(results) if results else 0,
|
||||||
|
"ttft_p50_sec": percentile(ttft_values, 50),
|
||||||
|
"ttft_p90_sec": percentile(ttft_values, 90),
|
||||||
|
"output_tps_p10_per_request": percentile(tps_values, 10),
|
||||||
|
"output_tps_p50_per_request": percentile(tps_values, 50),
|
||||||
|
"aggregate_output_tps": completion / wall if wall > 0 else 0,
|
||||||
|
"prompt_tokens": prompt,
|
||||||
|
"cached_tokens": cached,
|
||||||
|
"completion_tokens": completion,
|
||||||
|
"reasoning_tokens": sum(r.reasoning_tokens for r in ok),
|
||||||
|
"chars": sum(r.chars for r in ok),
|
||||||
|
"monitor": summarize_monitor(args.monitor_out) if args.monitor_out else None,
|
||||||
|
"results": [asdict(r) for r in results],
|
||||||
|
}
|
||||||
|
Path(args.out).parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
Path(args.out).write_text(json.dumps(summary, ensure_ascii=False, indent=2), encoding="utf-8")
|
||||||
|
print(json.dumps({k: v for k, v in summary.items() if k != "results"}, ensure_ascii=False, indent=2))
|
||||||
|
print("RESULT_FILE", args.out)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
255
worklogs/formal_perf_bench.py
Normal file
255
worklogs/formal_perf_bench.py
Normal file
@@ -0,0 +1,255 @@
|
|||||||
|
import argparse
|
||||||
|
import concurrent.futures
|
||||||
|
import json
|
||||||
|
import math
|
||||||
|
import statistics
|
||||||
|
import time
|
||||||
|
import urllib.error
|
||||||
|
import urllib.request
|
||||||
|
from dataclasses import dataclass, asdict
|
||||||
|
from datetime import datetime
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class RequestResult:
|
||||||
|
ok: bool
|
||||||
|
elapsed_sec: float
|
||||||
|
ttft_sec: float | None
|
||||||
|
prompt_tokens: int
|
||||||
|
cached_tokens: int
|
||||||
|
completion_tokens: int
|
||||||
|
reasoning_tokens: int
|
||||||
|
output_tps: float | None
|
||||||
|
error: str | None = None
|
||||||
|
|
||||||
|
|
||||||
|
def percentile(values, pct):
|
||||||
|
if not values:
|
||||||
|
return None
|
||||||
|
values = sorted(values)
|
||||||
|
if len(values) == 1:
|
||||||
|
return values[0]
|
||||||
|
pos = (len(values) - 1) * pct / 100.0
|
||||||
|
lo = math.floor(pos)
|
||||||
|
hi = math.ceil(pos)
|
||||||
|
if lo == hi:
|
||||||
|
return values[lo]
|
||||||
|
return values[lo] * (hi - pos) + values[hi] * (pos - lo)
|
||||||
|
|
||||||
|
|
||||||
|
def load_dataset(path):
|
||||||
|
text = Path(path).read_text(encoding="utf-8")
|
||||||
|
if path.endswith(".jsonl"):
|
||||||
|
data = [json.loads(line) for line in text.splitlines() if line.strip()]
|
||||||
|
else:
|
||||||
|
data = json.loads(text)
|
||||||
|
|
||||||
|
# Official-like format: each item is already an OpenAI chat completion
|
||||||
|
# request with messages/tools/tool_choice/stream/etc.
|
||||||
|
if data and isinstance(data[0], dict) and "messages" in data[0]:
|
||||||
|
return data
|
||||||
|
|
||||||
|
requests = []
|
||||||
|
for item in data:
|
||||||
|
system_prompt = item.get("system_prompt") or "You are a helpful assistant."
|
||||||
|
history = [{"role": "system", "content": system_prompt}]
|
||||||
|
for question in item.get("user_questions", []):
|
||||||
|
messages = history + [{"role": "user", "content": question}]
|
||||||
|
requests.append({"messages": messages})
|
||||||
|
# Synthetic assistant placeholder keeps later prompts multi-turn.
|
||||||
|
history = messages + [{"role": "assistant", "content": "好的,我们继续。"}]
|
||||||
|
return requests
|
||||||
|
|
||||||
|
|
||||||
|
def post_stream(url, model, request_item, max_tokens, timeout):
|
||||||
|
allowed_fields = {
|
||||||
|
"messages",
|
||||||
|
"tools",
|
||||||
|
"tool_choice",
|
||||||
|
"model",
|
||||||
|
"max_tokens",
|
||||||
|
"temperature",
|
||||||
|
"top_p",
|
||||||
|
"stop",
|
||||||
|
"presence_penalty",
|
||||||
|
"frequency_penalty",
|
||||||
|
"n",
|
||||||
|
"response_format",
|
||||||
|
}
|
||||||
|
payload = {k: v for k, v in dict(request_item).items() if k in allowed_fields}
|
||||||
|
payload["model"] = payload.get("model") or model
|
||||||
|
# The benchmark CLI controls output length, even if the synthetic dataset
|
||||||
|
# stores a larger official-like max_tokens value.
|
||||||
|
payload["max_tokens"] = max_tokens
|
||||||
|
payload["temperature"] = payload.get("temperature", 0)
|
||||||
|
payload["stream"] = True
|
||||||
|
payload["stream_options"] = {"include_usage": True}
|
||||||
|
req = urllib.request.Request(
|
||||||
|
url.rstrip("/") + "/v1/chat/completions",
|
||||||
|
data=json.dumps(payload, ensure_ascii=False).encode("utf-8"),
|
||||||
|
headers={"Content-Type": "application/json"},
|
||||||
|
method="POST",
|
||||||
|
)
|
||||||
|
|
||||||
|
start = time.perf_counter()
|
||||||
|
first_token_at = None
|
||||||
|
usage = {}
|
||||||
|
try:
|
||||||
|
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
||||||
|
for raw in resp:
|
||||||
|
line = raw.decode("utf-8", errors="replace").strip()
|
||||||
|
if not line or not line.startswith("data:"):
|
||||||
|
continue
|
||||||
|
data = line[5:].strip()
|
||||||
|
if data == "[DONE]":
|
||||||
|
break
|
||||||
|
obj = json.loads(data)
|
||||||
|
if obj.get("usage"):
|
||||||
|
usage = obj["usage"]
|
||||||
|
for choice in obj.get("choices") or []:
|
||||||
|
delta = choice.get("delta") or {}
|
||||||
|
text = delta.get("content") or delta.get("reasoning_content") or ""
|
||||||
|
if text and first_token_at is None:
|
||||||
|
first_token_at = time.perf_counter()
|
||||||
|
elapsed = time.perf_counter() - start
|
||||||
|
ttft = first_token_at - start if first_token_at else None
|
||||||
|
completion_tokens = int(usage.get("completion_tokens") or 0)
|
||||||
|
prompt_tokens = int(usage.get("prompt_tokens") or 0)
|
||||||
|
reasoning_tokens = int(usage.get("reasoning_tokens") or 0)
|
||||||
|
details = usage.get("prompt_tokens_details") or {}
|
||||||
|
cached_tokens = int(details.get("cached_tokens") or 0)
|
||||||
|
decode_sec = elapsed - ttft if ttft is not None else elapsed
|
||||||
|
output_tps = completion_tokens / decode_sec if completion_tokens and decode_sec > 0 else None
|
||||||
|
return RequestResult(
|
||||||
|
ok=True,
|
||||||
|
elapsed_sec=elapsed,
|
||||||
|
ttft_sec=ttft,
|
||||||
|
prompt_tokens=prompt_tokens,
|
||||||
|
cached_tokens=cached_tokens,
|
||||||
|
completion_tokens=completion_tokens,
|
||||||
|
reasoning_tokens=reasoning_tokens,
|
||||||
|
output_tps=output_tps,
|
||||||
|
)
|
||||||
|
except urllib.error.HTTPError as exc:
|
||||||
|
body = ""
|
||||||
|
try:
|
||||||
|
body = exc.read().decode("utf-8", errors="replace")
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
elapsed = time.perf_counter() - start
|
||||||
|
return RequestResult(
|
||||||
|
ok=False,
|
||||||
|
elapsed_sec=elapsed,
|
||||||
|
ttft_sec=None,
|
||||||
|
prompt_tokens=0,
|
||||||
|
cached_tokens=0,
|
||||||
|
completion_tokens=0,
|
||||||
|
reasoning_tokens=0,
|
||||||
|
output_tps=None,
|
||||||
|
error=f"HTTPError {exc.code}: {body[:1000]}",
|
||||||
|
)
|
||||||
|
except Exception as exc:
|
||||||
|
elapsed = time.perf_counter() - start
|
||||||
|
return RequestResult(
|
||||||
|
ok=False,
|
||||||
|
elapsed_sec=elapsed,
|
||||||
|
ttft_sec=None,
|
||||||
|
prompt_tokens=0,
|
||||||
|
cached_tokens=0,
|
||||||
|
completion_tokens=0,
|
||||||
|
reasoning_tokens=0,
|
||||||
|
output_tps=None,
|
||||||
|
error=f"{type(exc).__name__}: {exc}",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
parser = argparse.ArgumentParser()
|
||||||
|
parser.add_argument("--url", default="http://127.0.0.1:1111")
|
||||||
|
parser.add_argument("--model", default="llm")
|
||||||
|
parser.add_argument("--dataset", required=True)
|
||||||
|
parser.add_argument("--concurrency", type=int, default=1)
|
||||||
|
parser.add_argument("--max-requests", type=int, default=20)
|
||||||
|
parser.add_argument("--max-tokens", type=int, default=128)
|
||||||
|
parser.add_argument("--timeout", type=int, default=600)
|
||||||
|
parser.add_argument("--out", default="/root/work/logs/formal_perf_result.json")
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
reqs = load_dataset(args.dataset)
|
||||||
|
if not reqs:
|
||||||
|
raise SystemExit("empty dataset")
|
||||||
|
scheduled = [reqs[i % len(reqs)] for i in range(args.max_requests)]
|
||||||
|
|
||||||
|
wall_start = time.perf_counter()
|
||||||
|
results = []
|
||||||
|
with concurrent.futures.ThreadPoolExecutor(max_workers=args.concurrency) as pool:
|
||||||
|
futures = [
|
||||||
|
pool.submit(post_stream, args.url, args.model, request_item, args.max_tokens, args.timeout)
|
||||||
|
for request_item in scheduled
|
||||||
|
]
|
||||||
|
for fut in concurrent.futures.as_completed(futures):
|
||||||
|
results.append(fut.result())
|
||||||
|
print(f"done {len(results)}/{len(scheduled)} ok={results[-1].ok}", flush=True)
|
||||||
|
wall_sec = time.perf_counter() - wall_start
|
||||||
|
|
||||||
|
ok_results = [r for r in results if r.ok]
|
||||||
|
success_rate = len(ok_results) / len(results) if results else 0.0
|
||||||
|
ttfts = [r.ttft_sec for r in ok_results if r.ttft_sec is not None]
|
||||||
|
output_tps_values = [r.output_tps for r in ok_results if r.output_tps is not None]
|
||||||
|
|
||||||
|
prompt_tokens = sum(r.prompt_tokens for r in ok_results)
|
||||||
|
cached_tokens = sum(r.cached_tokens for r in ok_results)
|
||||||
|
input_tokens_uncached = max(prompt_tokens - cached_tokens, 0)
|
||||||
|
output_tokens = sum(r.completion_tokens for r in ok_results)
|
||||||
|
|
||||||
|
aggregate_output_tps = output_tokens / wall_sec if wall_sec > 0 else 0.0
|
||||||
|
aggregate_input_tps = input_tokens_uncached / wall_sec if wall_sec > 0 else 0.0
|
||||||
|
aggregate_cache_tps = cached_tokens / wall_sec if wall_sec > 0 else 0.0
|
||||||
|
weighted = aggregate_output_tps * 16.796 + aggregate_input_tps * 2.799 + aggregate_cache_tps * 0.56
|
||||||
|
cache_hit_rate = cached_tokens / prompt_tokens if prompt_tokens else 0.0
|
||||||
|
|
||||||
|
summary = {
|
||||||
|
"created_at": datetime.now().isoformat(timespec="seconds"),
|
||||||
|
"url": args.url,
|
||||||
|
"model": args.model,
|
||||||
|
"dataset": args.dataset,
|
||||||
|
"concurrency": args.concurrency,
|
||||||
|
"max_requests": args.max_requests,
|
||||||
|
"max_tokens": args.max_tokens,
|
||||||
|
"wall_sec": wall_sec,
|
||||||
|
"success_rate": success_rate,
|
||||||
|
"ttft_p90_sec": percentile(ttfts, 90),
|
||||||
|
"output_tps_p10_per_request": percentile(output_tps_values, 10),
|
||||||
|
"aggregate_output_tps": aggregate_output_tps,
|
||||||
|
"aggregate_input_tps_uncached": aggregate_input_tps,
|
||||||
|
"aggregate_cache_tps": aggregate_cache_tps,
|
||||||
|
"cache_hit_rate": cache_hit_rate,
|
||||||
|
"weighted_token_throughput": weighted,
|
||||||
|
"totals": {
|
||||||
|
"requests": len(results),
|
||||||
|
"success": len(ok_results),
|
||||||
|
"prompt_tokens": prompt_tokens,
|
||||||
|
"input_tokens_uncached": input_tokens_uncached,
|
||||||
|
"cached_tokens": cached_tokens,
|
||||||
|
"completion_tokens": output_tokens,
|
||||||
|
"reasoning_tokens": sum(r.reasoning_tokens for r in ok_results),
|
||||||
|
},
|
||||||
|
"targets": {
|
||||||
|
"output_tps_p10_per_request_gte_20": (percentile(output_tps_values, 10) or 0) >= 20,
|
||||||
|
"ttft_p90_lte_5": (percentile(ttfts, 90) or 999) <= 5,
|
||||||
|
"cache_hit_rate_gte_50pct": cache_hit_rate >= 0.5,
|
||||||
|
"success_rate_gte_99pct": success_rate >= 0.99,
|
||||||
|
"weighted_token_throughput_gte_8000": weighted >= 8000,
|
||||||
|
},
|
||||||
|
"results": [asdict(r) for r in results],
|
||||||
|
}
|
||||||
|
|
||||||
|
Path(args.out).parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
Path(args.out).write_text(json.dumps(summary, ensure_ascii=False, indent=2), encoding="utf-8")
|
||||||
|
print(json.dumps({k: v for k, v in summary.items() if k != "results"}, ensure_ascii=False, indent=2))
|
||||||
|
print("RESULT_FILE", args.out)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
413
worklogs/generate_official_like_dataset.py
Normal file
413
worklogs/generate_official_like_dataset.py
Normal file
@@ -0,0 +1,413 @@
|
|||||||
|
import argparse
|
||||||
|
import json
|
||||||
|
import random
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
|
||||||
|
TOOL_NAMES = [
|
||||||
|
"terminal_exec",
|
||||||
|
"process_list",
|
||||||
|
"read_file",
|
||||||
|
"edit_file",
|
||||||
|
"write_file",
|
||||||
|
"glob_files",
|
||||||
|
"grep_code",
|
||||||
|
"web_search",
|
||||||
|
"web_fetch",
|
||||||
|
"browser_open",
|
||||||
|
"mcp_codebase_memory_search",
|
||||||
|
"mcp_codebase_memory_graph",
|
||||||
|
"sessions_spawn",
|
||||||
|
"sessions_send",
|
||||||
|
"subagents_run",
|
||||||
|
"memory_search",
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
SYSTEM_PROMPT = """You are Codex, a coding agent. You help the user modify, test, and reason about code.
|
||||||
|
Follow the available tools carefully. Prefer precise shell commands, inspect files before editing,
|
||||||
|
and return concise progress updates. When tool calls are needed, produce valid tool call arguments."""
|
||||||
|
|
||||||
|
|
||||||
|
COMMON_PROJECT_CONTEXT = """
|
||||||
|
Repository: /home/user/workspace/project
|
||||||
|
Task: implement features, inspect logs, run tests, patch code, summarize results.
|
||||||
|
Conventions:
|
||||||
|
- Use rg for search.
|
||||||
|
- Avoid destructive commands.
|
||||||
|
- Keep a worklog.
|
||||||
|
- Preserve unrelated user changes.
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
CODE_SNIPPET = """
|
||||||
|
def fetch_prices(symbols, start_date, end_date, retries=3):
|
||||||
|
results = []
|
||||||
|
for symbol in symbols:
|
||||||
|
payload = {
|
||||||
|
"symbol": symbol,
|
||||||
|
"start": start_date,
|
||||||
|
"end": end_date,
|
||||||
|
"adjust": "qfq",
|
||||||
|
}
|
||||||
|
# retry network request with backoff
|
||||||
|
for attempt in range(retries):
|
||||||
|
try:
|
||||||
|
results.append(client.get("/prices", params=payload))
|
||||||
|
break
|
||||||
|
except TimeoutError:
|
||||||
|
time.sleep(1.5 * (attempt + 1))
|
||||||
|
return results
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
def make_tool(name):
|
||||||
|
if "terminal" in name or "process" in name:
|
||||||
|
props = {"command": {"type": "string", "description": "Shell command to execute."}}
|
||||||
|
required = ["command"]
|
||||||
|
elif "read" in name:
|
||||||
|
props = {"path": {"type": "string"}}
|
||||||
|
required = ["path"]
|
||||||
|
elif "edit" in name:
|
||||||
|
props = {
|
||||||
|
"file_path": {"type": "string"},
|
||||||
|
"old_string": {"type": "string"},
|
||||||
|
"new_string": {"type": "string"},
|
||||||
|
}
|
||||||
|
required = ["file_path", "old_string", "new_string"]
|
||||||
|
elif "web" in name:
|
||||||
|
props = {"query": {"type": "string"}, "count": {"type": "integer"}}
|
||||||
|
required = ["query"]
|
||||||
|
else:
|
||||||
|
props = {"input": {"type": "string"}, "limit": {"type": "integer"}}
|
||||||
|
required = ["input"]
|
||||||
|
return {
|
||||||
|
"type": "function",
|
||||||
|
"function": {
|
||||||
|
"name": name,
|
||||||
|
"description": f"Synthetic benchmark tool: {name}",
|
||||||
|
"parameters": {
|
||||||
|
"type": "object",
|
||||||
|
"properties": props,
|
||||||
|
"required": required,
|
||||||
|
},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def rand_text(rng, approx_tokens):
|
||||||
|
# Mostly ASCII/code-like text keeps the chars/token ratio simple enough for
|
||||||
|
# synthetic load generation. This is approximate; server usage is authoritative.
|
||||||
|
unit = (
|
||||||
|
"tool output line: status=ok path=/home/user/project/src/module.py "
|
||||||
|
"grep result includes function names, stack frames, JSON fields, "
|
||||||
|
"中文说明:这里包含工具返回、日志片段、代码上下文和转义字符。 "
|
||||||
|
+ CODE_SNIPPET.replace("\n", "\\n")
|
||||||
|
+ "\n"
|
||||||
|
)
|
||||||
|
chars = max(1, approx_tokens * 4)
|
||||||
|
return (unit * (chars // len(unit) + 1))[:chars]
|
||||||
|
|
||||||
|
|
||||||
|
def make_tool_call(rng, idx):
|
||||||
|
name = rng.choice(TOOL_NAMES)
|
||||||
|
if "terminal" in name:
|
||||||
|
args = {
|
||||||
|
"command": "cd /home/user/project && rg -n \"TODO|FIXME|error\" src tests 2>/dev/null | head -50"
|
||||||
|
}
|
||||||
|
elif "read" in name:
|
||||||
|
args = {"path": "/home/user/workspace/.agents/skills/data-analysis/SKILL.md"}
|
||||||
|
elif "edit" in name:
|
||||||
|
args = {
|
||||||
|
"file_path": "/home/user/scripts/fetch_prices.py",
|
||||||
|
"old_string": "# 全局频率限制器实例\n_rate_limiter = RateLimiter(min_interval=2.0, max_per_minute=20)",
|
||||||
|
"new_string": "# 全局频率限制器实例 - 根据配额上调\n_rate_limiter = RateLimiter(min_interval=1.5, max_per_minute=40)",
|
||||||
|
}
|
||||||
|
elif "web" in name:
|
||||||
|
args = {"query": "深圳 周末活动 推荐 2026年7月", "count": 10}
|
||||||
|
else:
|
||||||
|
args = {"input": "inspect project memory and summarize related files", "limit": 20}
|
||||||
|
return {
|
||||||
|
"id": f"call_{idx}_{rng.randrange(10**8)}",
|
||||||
|
"type": "function",
|
||||||
|
"function": {"name": name, "arguments": json.dumps(args, ensure_ascii=False)},
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def target_prompt_tokens(rng, max_prompt_tokens=None):
|
||||||
|
# Approximate official distribution: many 4K-16K, meaningful 32K+ tail,
|
||||||
|
# small number of 128K+ requests.
|
||||||
|
x = rng.random()
|
||||||
|
if x < 0.08:
|
||||||
|
value = rng.randint(1000, 4000)
|
||||||
|
return min(value, max_prompt_tokens) if max_prompt_tokens else value
|
||||||
|
if x < 0.50:
|
||||||
|
value = rng.randint(4000, 16000)
|
||||||
|
return min(value, max_prompt_tokens) if max_prompt_tokens else value
|
||||||
|
if x < 0.74:
|
||||||
|
value = rng.randint(16000, 32000)
|
||||||
|
return min(value, max_prompt_tokens) if max_prompt_tokens else value
|
||||||
|
if x < 0.95:
|
||||||
|
high = min(128000, max_prompt_tokens) if max_prompt_tokens else 128000
|
||||||
|
value = rng.randint(32000, max(32000, high))
|
||||||
|
return value
|
||||||
|
high = min(235000, max_prompt_tokens) if max_prompt_tokens else 235000
|
||||||
|
return rng.randint(128000, max(128000, high)) if high >= 128000 else high
|
||||||
|
|
||||||
|
|
||||||
|
def target_output_tokens(rng):
|
||||||
|
x = rng.random()
|
||||||
|
if x < 0.66:
|
||||||
|
return rng.randint(64, 256)
|
||||||
|
if x < 0.96:
|
||||||
|
return rng.randint(256, 2000)
|
||||||
|
return rng.randint(2000, 8192)
|
||||||
|
|
||||||
|
|
||||||
|
def build_request(rng, req_id, session_id, shared_prefix_tokens, total_prompt_tokens):
|
||||||
|
tool_count = max(1, int(rng.lognormvariate(2.8, 0.8)))
|
||||||
|
tool_count = min(tool_count, 92)
|
||||||
|
tools = [make_tool(rng.choice(TOOL_NAMES)) for _ in range(tool_count)]
|
||||||
|
|
||||||
|
messages = [
|
||||||
|
{
|
||||||
|
"role": "system",
|
||||||
|
"content": SYSTEM_PROMPT + "\n" + COMMON_PROJECT_CONTEXT + rand_text(rng, shared_prefix_tokens),
|
||||||
|
},
|
||||||
|
]
|
||||||
|
|
||||||
|
remaining = max(512, total_prompt_tokens - shared_prefix_tokens)
|
||||||
|
turns = rng.randint(12, 65)
|
||||||
|
per_turn = max(32, remaining // max(1, turns))
|
||||||
|
call_idx = 0
|
||||||
|
for turn in range(turns):
|
||||||
|
messages.append({
|
||||||
|
"role": "user",
|
||||||
|
"content": (
|
||||||
|
f"Session {session_id}, turn {turn}: inspect the codebase and continue the task. "
|
||||||
|
"Need exact commands, possible edits, and concise next action."
|
||||||
|
),
|
||||||
|
})
|
||||||
|
if rng.random() < 0.78:
|
||||||
|
calls = [make_tool_call(rng, call_idx)]
|
||||||
|
call_idx += 1
|
||||||
|
messages.append({
|
||||||
|
"role": "assistant",
|
||||||
|
# This runtime validates that every message has content or
|
||||||
|
# reasoning_content, even when assistant emits tool_calls.
|
||||||
|
"content": "",
|
||||||
|
"tool_calls": calls,
|
||||||
|
})
|
||||||
|
tool_len = int(per_turn * rng.uniform(0.5, 2.0))
|
||||||
|
if rng.random() > 0.99:
|
||||||
|
tool_len = rng.randint(12000, 50000)
|
||||||
|
elif rng.random() > 0.90:
|
||||||
|
tool_len = rng.randint(2500, 16000)
|
||||||
|
messages.append({
|
||||||
|
"role": "tool",
|
||||||
|
"tool_call_id": calls[0]["id"],
|
||||||
|
"name": calls[0]["function"]["name"],
|
||||||
|
"content": rand_text(rng, tool_len),
|
||||||
|
})
|
||||||
|
else:
|
||||||
|
messages.append({
|
||||||
|
"role": "assistant",
|
||||||
|
"content": "我会先检查相关文件和日志,再给出下一步修改建议。",
|
||||||
|
})
|
||||||
|
|
||||||
|
messages.append({
|
||||||
|
"role": "user",
|
||||||
|
"content": "基于以上工具结果,继续完成当前任务。需要时发起下一步 tool_call。",
|
||||||
|
})
|
||||||
|
|
||||||
|
return {
|
||||||
|
"id": f"synthetic_{req_id:04d}",
|
||||||
|
"model": "llm",
|
||||||
|
"messages": messages,
|
||||||
|
"tools": tools,
|
||||||
|
"tool_choice": "auto",
|
||||||
|
"stream": True,
|
||||||
|
"temperature": 0,
|
||||||
|
"max_tokens": target_output_tokens(rng),
|
||||||
|
"metadata": {
|
||||||
|
"session_id": session_id,
|
||||||
|
"target_prompt_tokens_approx": total_prompt_tokens,
|
||||||
|
"shared_prefix_tokens_approx": shared_prefix_tokens,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def estimate_tokens(messages, tools):
|
||||||
|
chars = 0
|
||||||
|
for msg in messages:
|
||||||
|
chars += len(msg.get("role") or "") + len(msg.get("content") or "")
|
||||||
|
if msg.get("tool_calls"):
|
||||||
|
chars += len(json.dumps(msg["tool_calls"], ensure_ascii=False))
|
||||||
|
if msg.get("name"):
|
||||||
|
chars += len(msg["name"])
|
||||||
|
chars += len(json.dumps(tools, ensure_ascii=False))
|
||||||
|
return max(1, chars // 4)
|
||||||
|
|
||||||
|
|
||||||
|
def make_history_turn(rng, state, turn_id, approx_tool_tokens):
|
||||||
|
user = {
|
||||||
|
"role": "user",
|
||||||
|
"content": (
|
||||||
|
f"Continue session {state['session_id']} turn {turn_id}. "
|
||||||
|
"Inspect files, reason about logs, and call the next tool if needed."
|
||||||
|
),
|
||||||
|
}
|
||||||
|
call = make_tool_call(rng, turn_id)
|
||||||
|
assistant = {
|
||||||
|
"role": "assistant",
|
||||||
|
"content": "",
|
||||||
|
"tool_calls": [call],
|
||||||
|
}
|
||||||
|
tool = {
|
||||||
|
"role": "tool",
|
||||||
|
"tool_call_id": call["id"],
|
||||||
|
"name": call["function"]["name"],
|
||||||
|
"content": rand_text(rng, approx_tool_tokens),
|
||||||
|
}
|
||||||
|
return [user, assistant, tool]
|
||||||
|
|
||||||
|
|
||||||
|
def generate_cumulative_requests(rng, count, session_count, max_prompt_tokens):
|
||||||
|
"""Generate requests whose prefixes are actually reusable.
|
||||||
|
|
||||||
|
The first version of this generator only stored an intended cache ratio in
|
||||||
|
metadata. This mode makes the byte/text prefix identical across requests by
|
||||||
|
sharing tools/system content and by growing each session history in place.
|
||||||
|
"""
|
||||||
|
common_tools = [make_tool(name) for name in TOOL_NAMES]
|
||||||
|
# Keep a large, fixed global prefix. It imitates common system prompt,
|
||||||
|
# agent framework instructions, and mounted tool schemas.
|
||||||
|
common_system = {
|
||||||
|
"role": "system",
|
||||||
|
"content": SYSTEM_PROMPT + "\n" + COMMON_PROJECT_CONTEXT + rand_text(rng, 6000),
|
||||||
|
}
|
||||||
|
sessions = {}
|
||||||
|
session_order = []
|
||||||
|
requests = []
|
||||||
|
|
||||||
|
def get_session(sid):
|
||||||
|
if sid not in sessions:
|
||||||
|
sessions[sid] = {
|
||||||
|
"session_id": sid,
|
||||||
|
"messages": [common_system],
|
||||||
|
"turn_id": 0,
|
||||||
|
}
|
||||||
|
session_order.append(sid)
|
||||||
|
return sessions[sid]
|
||||||
|
|
||||||
|
recent = []
|
||||||
|
for req_id in range(count):
|
||||||
|
# For small benchmark sets, force meaningful reuse. For larger sets,
|
||||||
|
# this still creates short-distance affinity similar to the official
|
||||||
|
# median gap of a few requests.
|
||||||
|
if recent and rng.random() < 0.65:
|
||||||
|
sid = rng.choice(recent[-32:])
|
||||||
|
else:
|
||||||
|
sid = f"sess_{len(session_order) % max(1, session_count):04d}"
|
||||||
|
recent.append(sid)
|
||||||
|
state = get_session(sid)
|
||||||
|
target = target_prompt_tokens(rng, max_prompt_tokens)
|
||||||
|
|
||||||
|
# Grow the historical context until this request has the target scale.
|
||||||
|
while estimate_tokens(state["messages"], common_tools) < max(1024, target - 256):
|
||||||
|
remaining = target - estimate_tokens(state["messages"], common_tools)
|
||||||
|
chunk = min(max(256, remaining), rng.randint(800, 3500))
|
||||||
|
state["messages"].extend(make_history_turn(rng, state, state["turn_id"], chunk))
|
||||||
|
state["turn_id"] += 1
|
||||||
|
|
||||||
|
current_user = {
|
||||||
|
"role": "user",
|
||||||
|
"content": "基于以上工具结果,继续完成当前任务。需要时发起下一步 tool_call。",
|
||||||
|
}
|
||||||
|
req_messages = list(state["messages"]) + [current_user]
|
||||||
|
requests.append({
|
||||||
|
"id": f"synthetic_cumulative_{req_id:04d}",
|
||||||
|
"model": "llm",
|
||||||
|
"messages": req_messages,
|
||||||
|
"tools": common_tools,
|
||||||
|
"tool_choice": "auto",
|
||||||
|
"stream": True,
|
||||||
|
"temperature": 0,
|
||||||
|
"max_tokens": target_output_tokens(rng),
|
||||||
|
"metadata": {
|
||||||
|
"session_id": sid,
|
||||||
|
"target_prompt_tokens_approx": estimate_tokens(req_messages, common_tools),
|
||||||
|
},
|
||||||
|
})
|
||||||
|
|
||||||
|
# Pretend the model made a tool call and the environment returned a
|
||||||
|
# result, so the next request in this session contains this full prefix.
|
||||||
|
state["messages"].append(current_user)
|
||||||
|
state["messages"].extend(make_history_turn(rng, state, state["turn_id"], rng.randint(500, 2500))[1:])
|
||||||
|
state["turn_id"] += 1
|
||||||
|
|
||||||
|
return requests
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
parser = argparse.ArgumentParser()
|
||||||
|
parser.add_argument("--out", required=True)
|
||||||
|
parser.add_argument("--requests", type=int, default=881)
|
||||||
|
parser.add_argument("--sessions", type=int, default=768)
|
||||||
|
parser.add_argument("--seed", type=int, default=20260714)
|
||||||
|
parser.add_argument("--jsonl", action="store_true")
|
||||||
|
parser.add_argument("--cumulative", action="store_true", help="Generate exact reusable session prefixes.")
|
||||||
|
parser.add_argument(
|
||||||
|
"--max-prompt-tokens",
|
||||||
|
type=int,
|
||||||
|
default=None,
|
||||||
|
help="Cap synthetic prompt-token target. Use 90000 for a server started with --max-model-len 100000.",
|
||||||
|
)
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
rng = random.Random(args.seed)
|
||||||
|
if args.cumulative:
|
||||||
|
requests = generate_cumulative_requests(rng, args.requests, args.sessions, args.max_prompt_tokens)
|
||||||
|
else:
|
||||||
|
session_ids = [f"sess_{i:04d}" for i in range(args.sessions)]
|
||||||
|
# Bias toward near-neighbor session reuse by reusing some recent sessions.
|
||||||
|
requests = []
|
||||||
|
recent = []
|
||||||
|
for i in range(args.requests):
|
||||||
|
if recent and rng.random() < 0.22:
|
||||||
|
session_id = rng.choice(recent[-32:])
|
||||||
|
else:
|
||||||
|
session_id = session_ids[i % len(session_ids)]
|
||||||
|
recent.append(session_id)
|
||||||
|
total = target_prompt_tokens(rng, args.max_prompt_tokens)
|
||||||
|
# Official weighted cache hit is about 65.6%; request-level varies.
|
||||||
|
if total >= 128000:
|
||||||
|
cache_ratio = rng.uniform(0.1, 0.35)
|
||||||
|
elif total >= 16000:
|
||||||
|
cache_ratio = rng.uniform(0.55, 0.72)
|
||||||
|
else:
|
||||||
|
cache_ratio = rng.uniform(0.75, 0.92)
|
||||||
|
shared = int(total * cache_ratio)
|
||||||
|
requests.append(build_request(rng, i, session_id, shared, total))
|
||||||
|
|
||||||
|
out = Path(args.out)
|
||||||
|
out.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
if args.jsonl:
|
||||||
|
out.write_text("\n".join(json.dumps(x, ensure_ascii=False) for x in requests) + "\n", encoding="utf-8")
|
||||||
|
else:
|
||||||
|
out.write_text(json.dumps(requests, ensure_ascii=False, indent=2), encoding="utf-8")
|
||||||
|
|
||||||
|
total_prompt = sum(x["metadata"]["target_prompt_tokens_approx"] for x in requests)
|
||||||
|
total_shared = sum(x["metadata"].get("shared_prefix_tokens_approx", 0) for x in requests)
|
||||||
|
print(json.dumps({
|
||||||
|
"out": str(out),
|
||||||
|
"requests": len(requests),
|
||||||
|
"sessions": len({x["metadata"]["session_id"] for x in requests}),
|
||||||
|
"approx_prompt_tokens_avg": total_prompt / len(requests),
|
||||||
|
"approx_cache_ratio_weighted": total_shared / total_prompt if total_shared else None,
|
||||||
|
"cumulative": args.cumulative,
|
||||||
|
}, ensure_ascii=False, indent=2))
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
@@ -0,0 +1,75 @@
|
|||||||
|
{
|
||||||
|
"created_at": "2026-07-14T13:34:44",
|
||||||
|
"label": "moe_tokenwise_short_c2_t128_r4",
|
||||||
|
"url": "http://127.0.0.1:1111",
|
||||||
|
"model": "llm",
|
||||||
|
"prompt_mode": "short",
|
||||||
|
"with_tools": false,
|
||||||
|
"tool_count": 0,
|
||||||
|
"concurrency": 2,
|
||||||
|
"requests": 4,
|
||||||
|
"max_tokens": 128,
|
||||||
|
"wall_sec": 45.37405508942902,
|
||||||
|
"success_rate": 1.0,
|
||||||
|
"ttft_p50_sec": 2.9874187149107456,
|
||||||
|
"ttft_p90_sec": 4.485993221774697,
|
||||||
|
"output_tps_p10_per_request": 6.428466023017078,
|
||||||
|
"output_tps_p50_per_request": 6.498744495352444,
|
||||||
|
"aggregate_output_tps": 11.283981539469739,
|
||||||
|
"prompt_tokens": 156,
|
||||||
|
"cached_tokens": 64,
|
||||||
|
"completion_tokens": 512,
|
||||||
|
"reasoning_tokens": 512,
|
||||||
|
"chars": 1841,
|
||||||
|
"monitor": null,
|
||||||
|
"results": [
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 23.971454864367843,
|
||||||
|
"ttft_sec": 4.486160388216376,
|
||||||
|
"completion_tokens": 128,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 0,
|
||||||
|
"reasoning_tokens": 128,
|
||||||
|
"output_tps": 6.569056482911376,
|
||||||
|
"chars": 449,
|
||||||
|
"error": null
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 23.97104039043188,
|
||||||
|
"ttft_sec": 4.485603166744113,
|
||||||
|
"completion_tokens": 128,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 0,
|
||||||
|
"reasoning_tokens": 128,
|
||||||
|
"output_tps": 6.569008358939714,
|
||||||
|
"chars": 449,
|
||||||
|
"error": null
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 21.4003098718822,
|
||||||
|
"ttft_sec": 1.4888528920710087,
|
||||||
|
"completion_tokens": 128,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 32,
|
||||||
|
"reasoning_tokens": 128,
|
||||||
|
"output_tps": 6.428459762125039,
|
||||||
|
"chars": 457,
|
||||||
|
"error": null
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 21.40062660165131,
|
||||||
|
"ttft_sec": 1.4892342630773783,
|
||||||
|
"completion_tokens": 128,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 32,
|
||||||
|
"reasoning_tokens": 128,
|
||||||
|
"output_tps": 6.428480631765174,
|
||||||
|
"chars": 486,
|
||||||
|
"error": null
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
@@ -0,0 +1,75 @@
|
|||||||
|
{
|
||||||
|
"created_at": "2026-07-14T12:30:35",
|
||||||
|
"label": "tiny_batch_moe_short_c2_t128_r4",
|
||||||
|
"url": "http://127.0.0.1:1111",
|
||||||
|
"model": "llm",
|
||||||
|
"prompt_mode": "short",
|
||||||
|
"with_tools": false,
|
||||||
|
"tool_count": 0,
|
||||||
|
"concurrency": 2,
|
||||||
|
"requests": 4,
|
||||||
|
"max_tokens": 128,
|
||||||
|
"wall_sec": 48.280083537101746,
|
||||||
|
"success_rate": 1.0,
|
||||||
|
"ttft_p50_sec": 2.915760966949165,
|
||||||
|
"ttft_p90_sec": 4.415460329316557,
|
||||||
|
"output_tps_p10_per_request": 5.954912943833064,
|
||||||
|
"output_tps_p50_per_request": 6.032240452288079,
|
||||||
|
"aggregate_output_tps": 10.604786953331262,
|
||||||
|
"prompt_tokens": 156,
|
||||||
|
"cached_tokens": 64,
|
||||||
|
"completion_tokens": 512,
|
||||||
|
"reasoning_tokens": 512,
|
||||||
|
"chars": 1796,
|
||||||
|
"monitor": null,
|
||||||
|
"results": [
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 25.909583542495966,
|
||||||
|
"ttft_sec": 4.414824679493904,
|
||||||
|
"completion_tokens": 128,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 0,
|
||||||
|
"reasoning_tokens": 128,
|
||||||
|
"output_tps": 5.95494003053556,
|
||||||
|
"chars": 449,
|
||||||
|
"error": null
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 25.91063128784299,
|
||||||
|
"ttft_sec": 4.415732750669122,
|
||||||
|
"completion_tokens": 128,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 0,
|
||||||
|
"reasoning_tokens": 128,
|
||||||
|
"output_tps": 5.954901335246281,
|
||||||
|
"chars": 449,
|
||||||
|
"error": null
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 22.367535073310137,
|
||||||
|
"ttft_sec": 1.4166972544044256,
|
||||||
|
"completion_tokens": 128,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 32,
|
||||||
|
"reasoning_tokens": 128,
|
||||||
|
"output_tps": 6.109540874040597,
|
||||||
|
"chars": 449,
|
||||||
|
"error": null
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 22.36741546355188,
|
||||||
|
"ttft_sec": 1.4165842793881893,
|
||||||
|
"completion_tokens": 128,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 32,
|
||||||
|
"reasoning_tokens": 128,
|
||||||
|
"output_tps": 6.109542808819567,
|
||||||
|
"chars": 449,
|
||||||
|
"error": null
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
@@ -0,0 +1,39 @@
|
|||||||
|
{
|
||||||
|
"created_at": "2026-07-14T11:39:06",
|
||||||
|
"label": "profile_eager_custom_ar_short_c1_t48_r1",
|
||||||
|
"url": "http://127.0.0.1:1111",
|
||||||
|
"model": "llm",
|
||||||
|
"prompt_mode": "short",
|
||||||
|
"with_tools": false,
|
||||||
|
"tool_count": 0,
|
||||||
|
"concurrency": 1,
|
||||||
|
"requests": 1,
|
||||||
|
"max_tokens": 48,
|
||||||
|
"wall_sec": 11.252297107130289,
|
||||||
|
"success_rate": 1.0,
|
||||||
|
"ttft_p50_sec": 4.141036370769143,
|
||||||
|
"ttft_p90_sec": 4.141036370769143,
|
||||||
|
"output_tps_p10_per_request": 6.7517276250194165,
|
||||||
|
"output_tps_p50_per_request": 6.7517276250194165,
|
||||||
|
"aggregate_output_tps": 4.265795645369481,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 0,
|
||||||
|
"completion_tokens": 48,
|
||||||
|
"reasoning_tokens": 48,
|
||||||
|
"chars": 173,
|
||||||
|
"monitor": null,
|
||||||
|
"results": [
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 11.250327898189425,
|
||||||
|
"ttft_sec": 4.141036370769143,
|
||||||
|
"completion_tokens": 48,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 0,
|
||||||
|
"reasoning_tokens": 48,
|
||||||
|
"output_tps": 6.7517276250194165,
|
||||||
|
"chars": 173,
|
||||||
|
"error": null
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
@@ -0,0 +1,51 @@
|
|||||||
|
{
|
||||||
|
"created_at": "2026-07-14T11:39:16",
|
||||||
|
"label": "profile_eager_custom_ar_short_c2_t32_r2",
|
||||||
|
"url": "http://127.0.0.1:1111",
|
||||||
|
"model": "llm",
|
||||||
|
"prompt_mode": "short",
|
||||||
|
"with_tools": false,
|
||||||
|
"tool_count": 0,
|
||||||
|
"concurrency": 2,
|
||||||
|
"requests": 2,
|
||||||
|
"max_tokens": 32,
|
||||||
|
"wall_sec": 10.011093640699983,
|
||||||
|
"success_rate": 1.0,
|
||||||
|
"ttft_p50_sec": 1.5351156890392303,
|
||||||
|
"ttft_p90_sec": 1.5355193987488747,
|
||||||
|
"output_tps_p10_per_request": 3.776407383659353,
|
||||||
|
"output_tps_p50_per_request": 3.7764316482081455,
|
||||||
|
"aggregate_output_tps": 6.392907937631185,
|
||||||
|
"prompt_tokens": 78,
|
||||||
|
"cached_tokens": 64,
|
||||||
|
"completion_tokens": 64,
|
||||||
|
"reasoning_tokens": 64,
|
||||||
|
"chars": 210,
|
||||||
|
"monitor": null,
|
||||||
|
"results": [
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 10.00828673131764,
|
||||||
|
"ttft_sec": 1.534611051902175,
|
||||||
|
"completion_tokens": 32,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 32,
|
||||||
|
"reasoning_tokens": 32,
|
||||||
|
"output_tps": 3.7764013175221547,
|
||||||
|
"chars": 105,
|
||||||
|
"error": null
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 10.00915989279747,
|
||||||
|
"ttft_sec": 1.5356203261762857,
|
||||||
|
"completion_tokens": 32,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 32,
|
||||||
|
"reasoning_tokens": 32,
|
||||||
|
"output_tps": 3.7764619788941363,
|
||||||
|
"chars": 105,
|
||||||
|
"error": null
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
@@ -0,0 +1,39 @@
|
|||||||
|
{
|
||||||
|
"created_at": "2026-07-14T12:02:31",
|
||||||
|
"label": "profile_mode_eager_custom_ar_short_c1_t24_r1",
|
||||||
|
"url": "http://127.0.0.1:1111",
|
||||||
|
"model": "llm",
|
||||||
|
"prompt_mode": "short",
|
||||||
|
"with_tools": false,
|
||||||
|
"tool_count": 0,
|
||||||
|
"concurrency": 1,
|
||||||
|
"requests": 1,
|
||||||
|
"max_tokens": 24,
|
||||||
|
"wall_sec": 7.6276699639856815,
|
||||||
|
"success_rate": 1.0,
|
||||||
|
"ttft_p50_sec": 4.090665258467197,
|
||||||
|
"ttft_p90_sec": 4.090665258467197,
|
||||||
|
"output_tps_p10_per_request": 6.788945343504622,
|
||||||
|
"output_tps_p50_per_request": 6.788945343504622,
|
||||||
|
"aggregate_output_tps": 3.1464392289279512,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 0,
|
||||||
|
"completion_tokens": 24,
|
||||||
|
"reasoning_tokens": 24,
|
||||||
|
"chars": 78,
|
||||||
|
"monitor": null,
|
||||||
|
"results": [
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 7.625824077054858,
|
||||||
|
"ttft_sec": 4.090665258467197,
|
||||||
|
"completion_tokens": 24,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 0,
|
||||||
|
"reasoning_tokens": 24,
|
||||||
|
"output_tps": 6.788945343504622,
|
||||||
|
"chars": 78,
|
||||||
|
"error": null
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
@@ -0,0 +1,301 @@
|
|||||||
|
/usr/local/corex/lib/python3/dist-packages/torch/cuda/__init__.py:51: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you.
|
||||||
|
import pynvml # type: ignore[import]
|
||||||
|
INFO 07-14 13:29:12 importing.py:10] Triton not installed; certain GPU-related functions will not be available.
|
||||||
|
2026-07-14 13:29:13.750824: I tensorflow/core/util/port.cc:110] oneDNN custom operations are on. You may see slightly different numerical results due to floating-point round-off errors from different computation orders. To turn them off, set the environment variable `TF_ENABLE_ONEDNN_OPTS=0`.
|
||||||
|
2026-07-14 13:29:13.802713: I tensorflow/core/platform/cpu_feature_guard.cc:182] This TensorFlow binary is optimized to use available CPU instructions in performance-critical operations.
|
||||||
|
To enable the following instructions: SSE3 SSE4.1 SSE4.2 AVX AVX2 AVX512F AVX512_VNNI AVX512_BF16 AVX_VNNI AMX_TILE AMX_INT8 AMX_BF16 FMA, in other operations, rebuild TensorFlow with the appropriate compiler flags.
|
||||||
|
WARNING:tensorflow:Deprecation warnings have been disabled. Set TF_ENABLE_DEPRECATION_WARNINGS=1 to re-enable them.
|
||||||
|
INFO 07-14 13:29:19 api_server.py:530] vLLM API server version 0.6.3
|
||||||
|
INFO 07-14 13:29:19 api_server.py:531] args: Namespace(host='0.0.0.0', port=1111, uvicorn_log_level='info', allow_credentials=False, allowed_origins=['*'], allowed_methods=['*'], allowed_headers=['*'], api_key=None, lora_modules=None, prompt_adapters=None, chat_template=None, response_role='assistant', ssl_keyfile=None, ssl_certfile=None, ssl_ca_certs=None, ssl_cert_reqs=0, root_path=None, middleware=[], return_tokens_as_token_ids=False, disable_frontend_multiprocessing=True, enable_auto_tool_choice=True, tool_call_parser='qwen3_coder', tool_parser_plugin='', reasoning_parser='qwen3', model='/root/public-storage/models/Qwen/Qwen3.6-35B-A3B', tokenizer=None, skip_tokenizer_init=False, revision=None, code_revision=None, tokenizer_revision=None, tokenizer_mode='auto', trust_remote_code=True, download_dir=None, load_format='auto', config_format='auto', dtype='auto', kv_cache_dtype='auto', quantization_param_path=None, max_model_len=100000, guided_decoding_backend='outlines', distributed_executor_backend=None, worker_use_ray=False, pipeline_parallel_size=1, tensor_parallel_size=4, max_parallel_loading_workers=None, ray_workers_use_nsight=False, block_size=16, enable_prefix_caching=True, disable_sliding_window=False, use_v2_block_manager=True, num_lookahead_slots=0, seed=0, swap_space=4, cpu_offload_gb=0, gpu_memory_utilization=0.95, num_gpu_blocks_override=None, max_num_batched_tokens=8192, max_num_seqs=2, max_logprobs=20, disable_log_stats=False, quantization=None, rope_scaling=None, rope_theta=None, enforce_eager=True, max_context_len_to_capture=None, max_seq_len_to_capture=32768, disable_custom_all_reduce=False, tokenizer_pool_size=0, tokenizer_pool_type='ray', tokenizer_pool_extra_config=None, limit_mm_per_prompt=None, mm_processor_kwargs=None, enable_lora=False, max_loras=1, max_lora_rank=16, lora_extra_vocab_size=256, lora_dtype='auto', long_lora_scaling_factors=None, max_cpu_loras=None, fully_sharded_loras=False, enable_prompt_adapter=False, max_prompt_adapters=1, max_prompt_adapter_token=0, device='auto', num_scheduler_steps=1, multi_step_stream_outputs=True, scheduler_delay_factor=0.0, enable_chunked_prefill=True, speculative_model=None, speculative_model_quantization=None, num_speculative_tokens=None, speculative_disable_mqa_scorer=False, speculative_draft_tensor_parallel_size=None, speculative_max_model_len=None, speculative_disable_by_batch_size=None, ngram_prompt_lookup_max=None, ngram_prompt_lookup_min=None, spec_decoding_acceptance_method='rejection_sampler', typical_acceptance_sampler_posterior_threshold=None, typical_acceptance_sampler_posterior_alpha=None, disable_logprobs_during_spec_decoding=None, model_loader_extra_config=None, ignore_patterns=[], preemption_mode=None, served_model_name=['llm'], qlora_adapter_name_or_path=None, otlp_traces_endpoint=None, collect_detailed_traces=None, disable_async_output_proc=False, override_neuron_config=None, scheduling_policy='fcfs', disable_log_requests=True, max_log_len=None, disable_fastapi_docs=False)
|
||||||
|
INFO 07-14 13:29:19 config.py:1670] Downcasting torch.float32 to torch.float16.
|
||||||
|
INFO 07-14 13:29:30 config.py:887] Defaulting to use mp for distributed inference
|
||||||
|
INFO 07-14 13:29:30 config.py:1005] Chunked prefill is enabled with max_num_batched_tokens=8192.
|
||||||
|
WARNING 07-14 13:29:30 config.py:380] To see benefits of async output processing, enable CUDA graph. Since, enforce-eager is enabled, async output processor cannot be used
|
||||||
|
INFO 07-14 13:29:30 llm_engine.py:237] Initializing an LLM engine (v0.6.3) with config: model='/root/public-storage/models/Qwen/Qwen3.6-35B-A3B', speculative_config=None, tokenizer='/root/public-storage/models/Qwen/Qwen3.6-35B-A3B', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, override_neuron_config=None, rope_scaling=None, rope_theta=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.float16, max_seq_len=100000, download_dir=None, load_format=LoadFormat.AUTO, tensor_parallel_size=4, pipeline_parallel_size=1, disable_custom_all_reduce=False, quantization=None, enforce_eager=True, kv_cache_dtype=auto, quantization_param_path=None, device_config=cuda, decoding_config=DecodingConfig(guided_decoding_backend='outlines'), observability_config=ObservabilityConfig(otlp_traces_endpoint=None, collect_model_forward_time=False, collect_model_execute_time=False), seed=0, served_model_name=llm, use_v2_block_manager=True, num_scheduler_steps=1, chunked_prefill_enabled=True multi_step_stream_outputs=True, enable_prefix_caching=True, use_async_output_proc=False, use_cached_outputs=False, mm_processor_kwargs=None)
|
||||||
|
WARNING 07-14 13:29:30 multiproc_gpu_executor.py:53] Reducing Torch parallelism from 64 threads to 1 to avoid unnecessary CPU contention. Set OMP_NUM_THREADS in the external environment to tune this value as needed.
|
||||||
|
INFO 07-14 13:29:30 custom_cache_manager.py:17] Setting Triton cache manager to: vllm.triton_utils.custom_cache_manager:CustomCacheManager
|
||||||
|
INFO 07-14 13:29:30 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
INFO 07-14 13:29:30 selector.py:115] Using XFormers backend.
|
||||||
|
/usr/local/corex/lib/python3/dist-packages/torch/cuda/__init__.py:51: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you.
|
||||||
|
import pynvml # type: ignore[import]
|
||||||
|
/usr/local/corex/lib/python3/dist-packages/torch/cuda/__init__.py:51: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you.
|
||||||
|
import pynvml # type: ignore[import]
|
||||||
|
/usr/local/corex/lib/python3/dist-packages/torch/cuda/__init__.py:51: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you.
|
||||||
|
import pynvml # type: ignore[import]
|
||||||
|
INFO 07-14 13:29:32 importing.py:10] Triton not installed; certain GPU-related functions will not be available.
|
||||||
|
INFO 07-14 13:29:32 importing.py:10] Triton not installed; certain GPU-related functions will not be available.
|
||||||
|
INFO 07-14 13:29:32 importing.py:10] Triton not installed; certain GPU-related functions will not be available.
|
||||||
|
WARNING:tensorflow:Deprecation warnings have been disabled. Set TF_ENABLE_DEPRECATION_WARNINGS=1 to re-enable them.
|
||||||
|
WARNING:tensorflow:Deprecation warnings have been disabled. Set TF_ENABLE_DEPRECATION_WARNINGS=1 to re-enable them.
|
||||||
|
WARNING:tensorflow:Deprecation warnings have been disabled. Set TF_ENABLE_DEPRECATION_WARNINGS=1 to re-enable them.
|
||||||
|
[1;36m(VllmWorkerProcess pid=15585)[0;0m INFO 07-14 13:29:39 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=15585)[0;0m INFO 07-14 13:29:39 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=15585)[0;0m INFO 07-14 13:29:39 multiproc_worker_utils.py:216] Worker ready; awaiting tasks
|
||||||
|
[1;36m(VllmWorkerProcess pid=15584)[0;0m INFO 07-14 13:29:40 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=15584)[0;0m INFO 07-14 13:29:40 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=15584)[0;0m INFO 07-14 13:29:40 multiproc_worker_utils.py:216] Worker ready; awaiting tasks
|
||||||
|
[1;36m(VllmWorkerProcess pid=15583)[0;0m INFO 07-14 13:29:40 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=15583)[0;0m INFO 07-14 13:29:40 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=15583)[0;0m INFO 07-14 13:29:40 multiproc_worker_utils.py:216] Worker ready; awaiting tasks
|
||||||
|
INFO 07-14 13:29:40 shm_broadcast.py:242] vLLM message queue communication handle: Handle(connect_ip='127.0.0.1', local_reader_ranks=[1, 2, 3], buffer=<vllm.distributed.device_communicators.shm_broadcast.ShmRingBuffer object at 0x7fb7d24c1570>, local_subscribe_port=42675, remote_subscribe_port=None)
|
||||||
|
INFO 07-14 13:29:40 model_runner.py:1112] Starting to load model /root/public-storage/models/Qwen/Qwen3.6-35B-A3B...
|
||||||
|
[1;36m(VllmWorkerProcess pid=15583)[0;0m INFO 07-14 13:29:40 model_runner.py:1112] Starting to load model /root/public-storage/models/Qwen/Qwen3.6-35B-A3B...
|
||||||
|
[1;36m(VllmWorkerProcess pid=15584)[0;0m INFO 07-14 13:29:40 model_runner.py:1112] Starting to load model /root/public-storage/models/Qwen/Qwen3.6-35B-A3B...
|
||||||
|
[1;36m(VllmWorkerProcess pid=15585)[0;0m INFO 07-14 13:29:40 model_runner.py:1112] Starting to load model /root/public-storage/models/Qwen/Qwen3.6-35B-A3B...
|
||||||
|
INFO 07-14 13:29:40 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
INFO 07-14 13:29:40 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=15584)[0;0m INFO 07-14 13:29:40 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=15584)[0;0m INFO 07-14 13:29:40 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=15585)[0;0m INFO 07-14 13:29:40 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=15585)[0;0m INFO 07-14 13:29:40 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=15583)[0;0m INFO 07-14 13:29:40 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=15583)[0;0m INFO 07-14 13:29:40 selector.py:115] Using XFormers backend.
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 0% Completed | 0/26 [00:00<?, ?it/s]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 4% Completed | 1/26 [00:01<00:45, 1.80s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 8% Completed | 2/26 [00:02<00:23, 1.02it/s]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 12% Completed | 3/26 [00:03<00:30, 1.34s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 15% Completed | 4/26 [00:05<00:32, 1.49s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 19% Completed | 5/26 [00:06<00:26, 1.28s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 23% Completed | 6/26 [00:08<00:29, 1.48s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 27% Completed | 7/26 [00:08<00:21, 1.12s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 31% Completed | 8/26 [00:10<00:23, 1.29s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 35% Completed | 9/26 [00:11<00:19, 1.16s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 38% Completed | 10/26 [00:13<00:22, 1.40s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 42% Completed | 11/26 [00:13<00:15, 1.05s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 46% Completed | 12/26 [00:15<00:17, 1.27s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 50% Completed | 13/26 [00:17<00:18, 1.43s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 54% Completed | 14/26 [00:19<00:19, 1.59s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 58% Completed | 15/26 [00:19<00:14, 1.35s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 62% Completed | 16/26 [00:22<00:16, 1.64s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 65% Completed | 17/26 [00:22<00:11, 1.28s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 69% Completed | 18/26 [00:24<00:11, 1.42s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 73% Completed | 19/26 [00:25<00:10, 1.47s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 77% Completed | 20/26 [00:28<00:10, 1.72s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 81% Completed | 21/26 [00:29<00:07, 1.49s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 85% Completed | 22/26 [00:29<00:05, 1.26s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 88% Completed | 23/26 [00:32<00:04, 1.56s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 92% Completed | 24/26 [00:34<00:03, 1.63s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 96% Completed | 25/26 [00:35<00:01, 1.69s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 100% Completed | 26/26 [00:36<00:00, 1.33s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 100% Completed | 26/26 [00:36<00:00, 1.40s/it]
|
||||||
|
|
||||||
|
INFO 07-14 13:30:17 model_runner.py:1123] Loading model weights took 16.2303 GB
|
||||||
|
[1;36m(VllmWorkerProcess pid=15583)[0;0m INFO 07-14 13:30:21 model_runner.py:1123] Loading model weights took 16.2303 GB
|
||||||
|
[1;36m(VllmWorkerProcess pid=15585)[0;0m INFO 07-14 13:30:21 model_runner.py:1123] Loading model weights took 16.2303 GB
|
||||||
|
[1;36m(VllmWorkerProcess pid=15584)[0;0m INFO 07-14 13:30:21 model_runner.py:1123] Loading model weights took 16.2303 GB
|
||||||
|
INFO 07-14 13:30:29 distributed_gpu_executor.py:57] # GPU blocks: 21100, # CPU blocks: 6553
|
||||||
|
INFO 07-14 13:30:29 distributed_gpu_executor.py:61] Maximum concurrency for 100000 tokens per request: 3.38x
|
||||||
|
INFO 07-14 13:30:33 serving_chat.py:79] "auto" tool choice has been enabled please note that while the parallel_tool_calls client option is preset for compatibility reasons, it will be ignored.
|
||||||
|
INFO 07-14 13:30:33 serving_chat.py:101] Reasoning parser 'qwen3' enabled.
|
||||||
|
WARNING 07-14 13:30:33 serving_embedding.py:199] embedding_mode is False. Embedding API will not work.
|
||||||
|
INFO 07-14 13:30:33 launcher.py:19] Available routes are:
|
||||||
|
INFO 07-14 13:30:33 launcher.py:27] Route: /openapi.json, Methods: GET, HEAD
|
||||||
|
INFO 07-14 13:30:33 launcher.py:27] Route: /docs, Methods: GET, HEAD
|
||||||
|
INFO 07-14 13:30:33 launcher.py:27] Route: /docs/oauth2-redirect, Methods: GET, HEAD
|
||||||
|
INFO 07-14 13:30:33 launcher.py:27] Route: /redoc, Methods: GET, HEAD
|
||||||
|
INFO 07-14 13:30:33 launcher.py:27] Route: /health, Methods: GET
|
||||||
|
INFO 07-14 13:30:33 launcher.py:27] Route: /tokenize, Methods: POST
|
||||||
|
INFO 07-14 13:30:33 launcher.py:27] Route: /detokenize, Methods: POST
|
||||||
|
INFO 07-14 13:30:33 launcher.py:27] Route: /v1/models, Methods: GET
|
||||||
|
INFO 07-14 13:30:33 launcher.py:27] Route: /version, Methods: GET
|
||||||
|
INFO 07-14 13:30:33 launcher.py:27] Route: /v1/chat/completions, Methods: POST
|
||||||
|
INFO 07-14 13:30:33 launcher.py:27] Route: /v1/completions, Methods: POST
|
||||||
|
INFO 07-14 13:30:33 launcher.py:27] Route: /v1/embeddings, Methods: POST
|
||||||
|
INFO: Started server process [15244]
|
||||||
|
INFO: Waiting for application startup.
|
||||||
|
INFO: Application startup complete.
|
||||||
|
INFO: Uvicorn running on socket ('0.0.0.0', 1111) (Press CTRL+C to quit)
|
||||||
|
INFO 07-14 13:30:43 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:30:43 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:30:53 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:30:53 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:31:03 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:31:03 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:31:13 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:31:13 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:31:23 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:31:23 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:31:33 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:31:33 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:31:43 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:31:43 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:31:53 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:31:53 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:32:03 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:32:03 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO: 127.0.0.1:55020 - "GET /health HTTP/1.1" 200 OK
|
||||||
|
INFO 07-14 13:32:13 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:32:13 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:32:23 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:32:23 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:32:33 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:32:33 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:32:43 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:32:43 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:32:53 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:32:53 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:33:03 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:33:03 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:33:13 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:33:13 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO: 127.0.0.1:33690 - "GET /health HTTP/1.1" 200 OK
|
||||||
|
INFO 07-14 13:33:23 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:33:23 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:33:33 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:33:33 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:33:43 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:33:43 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:33:53 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:33:53 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO: 127.0.0.1:41050 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||||
|
INFO: 127.0.0.1:41064 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||||
|
/usr/local/lib/python3.10/site-packages/pyairports/airports.py:1: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81.
|
||||||
|
from pkg_resources import resource_string
|
||||||
|
INFO 07-14 13:34:03 metrics.py:345] Avg prompt throughput: 7.8 tokens/s, Avg generation throughput: 0.2 tokens/s, Running: 2 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:34:03 metrics.py:361] Prefix cache hit rate: GPU: 50.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:34:08 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 12.6 tokens/s, Running: 2 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:34:08 metrics.py:361] Prefix cache hit rate: GPU: 50.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:34:13 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 12.9 tokens/s, Running: 2 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:34:13 metrics.py:361] Prefix cache hit rate: GPU: 50.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:34:18 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 13.6 tokens/s, Running: 2 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:34:18 metrics.py:361] Prefix cache hit rate: GPU: 50.00%, CPU: 0.00%
|
||||||
|
INFO: 127.0.0.1:57588 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||||
|
INFO: 127.0.0.1:57592 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||||
|
INFO 07-14 13:34:24 metrics.py:345] Avg prompt throughput: 13.5 tokens/s, Avg generation throughput: 10.1 tokens/s, Running: 2 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:34:24 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:34:29 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 13.0 tokens/s, Running: 2 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:34:29 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:34:34 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 12.9 tokens/s, Running: 2 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:34:34 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:34:39 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 12.7 tokens/s, Running: 2 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:34:39 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:34:53 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 4.2 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:34:53 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:35:03 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:35:03 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:35:13 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:35:13 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:35:23 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:35:23 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:35:33 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:35:33 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:35:43 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:35:43 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:35:53 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:35:53 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:36:03 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:36:03 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:36:13 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:36:13 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:36:23 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:36:23 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:36:33 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:36:33 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:36:43 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:36:43 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:36:53 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:36:53 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:37:03 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:37:03 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:37:13 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:37:13 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:37:23 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:37:23 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:37:33 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:37:33 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:37:43 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:37:43 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:37:53 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:37:53 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:38:03 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:38:03 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:38:13 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:38:13 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:38:23 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:38:23 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:38:33 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:38:33 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:38:43 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:38:43 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:38:53 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:38:53 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:39:03 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:39:03 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:39:13 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:39:13 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:39:23 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:39:23 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:39:33 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:39:33 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:39:43 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:39:43 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:39:53 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:39:53 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:40:03 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:40:03 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:40:13 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:40:13 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:40:23 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:40:23 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:40:33 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:40:33 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:40:43 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:40:43 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:40:53 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:40:53 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:41:03 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:41:03 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:41:13 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:41:13 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:41:23 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:41:23 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:41:33 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:41:33 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:41:43 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:41:43 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:41:53 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:41:53 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:42:03 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:42:03 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:42:13 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:42:13 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:42:23 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:42:23 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:42:33 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:42:33 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:42:43 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:42:43 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:42:53 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:42:53 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:43:03 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:43:03 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 13:43:13 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 13:43:13 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
@@ -0,0 +1,407 @@
|
|||||||
|
/usr/local/corex/lib/python3/dist-packages/torch/cuda/__init__.py:51: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you.
|
||||||
|
import pynvml # type: ignore[import]
|
||||||
|
INFO 07-14 11:34:16 importing.py:10] Triton not installed; certain GPU-related functions will not be available.
|
||||||
|
2026-07-14 11:34:18.268441: I tensorflow/core/util/port.cc:110] oneDNN custom operations are on. You may see slightly different numerical results due to floating-point round-off errors from different computation orders. To turn them off, set the environment variable `TF_ENABLE_ONEDNN_OPTS=0`.
|
||||||
|
2026-07-14 11:34:18.320614: I tensorflow/core/platform/cpu_feature_guard.cc:182] This TensorFlow binary is optimized to use available CPU instructions in performance-critical operations.
|
||||||
|
To enable the following instructions: SSE3 SSE4.1 SSE4.2 AVX AVX2 AVX512F AVX512_VNNI AVX512_BF16 AVX_VNNI AMX_TILE AMX_INT8 AMX_BF16 FMA, in other operations, rebuild TensorFlow with the appropriate compiler flags.
|
||||||
|
WARNING:tensorflow:Deprecation warnings have been disabled. Set TF_ENABLE_DEPRECATION_WARNINGS=1 to re-enable them.
|
||||||
|
INFO 07-14 11:34:23 api_server.py:530] vLLM API server version 0.6.3
|
||||||
|
INFO 07-14 11:34:23 api_server.py:531] args: Namespace(host='0.0.0.0', port=1111, uvicorn_log_level='info', allow_credentials=False, allowed_origins=['*'], allowed_methods=['*'], allowed_headers=['*'], api_key=None, lora_modules=None, prompt_adapters=None, chat_template=None, response_role='assistant', ssl_keyfile=None, ssl_certfile=None, ssl_ca_certs=None, ssl_cert_reqs=0, root_path=None, middleware=[], return_tokens_as_token_ids=False, disable_frontend_multiprocessing=True, enable_auto_tool_choice=True, tool_call_parser='qwen3_coder', tool_parser_plugin='', reasoning_parser='qwen3', model='/root/public-storage/models/Qwen/Qwen3.6-35B-A3B', tokenizer=None, skip_tokenizer_init=False, revision=None, code_revision=None, tokenizer_revision=None, tokenizer_mode='auto', trust_remote_code=True, download_dir=None, load_format='auto', config_format='auto', dtype='auto', kv_cache_dtype='auto', quantization_param_path=None, max_model_len=100000, guided_decoding_backend='outlines', distributed_executor_backend=None, worker_use_ray=False, pipeline_parallel_size=1, tensor_parallel_size=4, max_parallel_loading_workers=None, ray_workers_use_nsight=False, block_size=16, enable_prefix_caching=True, disable_sliding_window=False, use_v2_block_manager=True, num_lookahead_slots=0, seed=0, swap_space=4, cpu_offload_gb=0, gpu_memory_utilization=0.95, num_gpu_blocks_override=None, max_num_batched_tokens=8192, max_num_seqs=2, max_logprobs=20, disable_log_stats=False, quantization=None, rope_scaling=None, rope_theta=None, enforce_eager=True, max_context_len_to_capture=None, max_seq_len_to_capture=32768, disable_custom_all_reduce=False, tokenizer_pool_size=0, tokenizer_pool_type='ray', tokenizer_pool_extra_config=None, limit_mm_per_prompt=None, mm_processor_kwargs=None, enable_lora=False, max_loras=1, max_lora_rank=16, lora_extra_vocab_size=256, lora_dtype='auto', long_lora_scaling_factors=None, max_cpu_loras=None, fully_sharded_loras=False, enable_prompt_adapter=False, max_prompt_adapters=1, max_prompt_adapter_token=0, device='auto', num_scheduler_steps=1, multi_step_stream_outputs=True, scheduler_delay_factor=0.0, enable_chunked_prefill=True, speculative_model=None, speculative_model_quantization=None, num_speculative_tokens=None, speculative_disable_mqa_scorer=False, speculative_draft_tensor_parallel_size=None, speculative_max_model_len=None, speculative_disable_by_batch_size=None, ngram_prompt_lookup_max=None, ngram_prompt_lookup_min=None, spec_decoding_acceptance_method='rejection_sampler', typical_acceptance_sampler_posterior_threshold=None, typical_acceptance_sampler_posterior_alpha=None, disable_logprobs_during_spec_decoding=None, model_loader_extra_config=None, ignore_patterns=[], preemption_mode=None, served_model_name=['llm'], qlora_adapter_name_or_path=None, otlp_traces_endpoint=None, collect_detailed_traces=None, disable_async_output_proc=False, override_neuron_config=None, scheduling_policy='fcfs', disable_log_requests=True, max_log_len=None, disable_fastapi_docs=False)
|
||||||
|
INFO 07-14 11:34:23 config.py:1670] Downcasting torch.float32 to torch.float16.
|
||||||
|
INFO 07-14 11:34:34 config.py:887] Defaulting to use mp for distributed inference
|
||||||
|
INFO 07-14 11:34:34 config.py:1005] Chunked prefill is enabled with max_num_batched_tokens=8192.
|
||||||
|
WARNING 07-14 11:34:34 config.py:380] To see benefits of async output processing, enable CUDA graph. Since, enforce-eager is enabled, async output processor cannot be used
|
||||||
|
INFO 07-14 11:34:34 llm_engine.py:237] Initializing an LLM engine (v0.6.3) with config: model='/root/public-storage/models/Qwen/Qwen3.6-35B-A3B', speculative_config=None, tokenizer='/root/public-storage/models/Qwen/Qwen3.6-35B-A3B', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, override_neuron_config=None, rope_scaling=None, rope_theta=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.float16, max_seq_len=100000, download_dir=None, load_format=LoadFormat.AUTO, tensor_parallel_size=4, pipeline_parallel_size=1, disable_custom_all_reduce=False, quantization=None, enforce_eager=True, kv_cache_dtype=auto, quantization_param_path=None, device_config=cuda, decoding_config=DecodingConfig(guided_decoding_backend='outlines'), observability_config=ObservabilityConfig(otlp_traces_endpoint=None, collect_model_forward_time=False, collect_model_execute_time=False), seed=0, served_model_name=llm, use_v2_block_manager=True, num_scheduler_steps=1, chunked_prefill_enabled=True multi_step_stream_outputs=True, enable_prefix_caching=True, use_async_output_proc=False, use_cached_outputs=False, mm_processor_kwargs=None)
|
||||||
|
WARNING 07-14 11:34:35 multiproc_gpu_executor.py:53] Reducing Torch parallelism from 64 threads to 1 to avoid unnecessary CPU contention. Set OMP_NUM_THREADS in the external environment to tune this value as needed.
|
||||||
|
INFO 07-14 11:34:35 custom_cache_manager.py:17] Setting Triton cache manager to: vllm.triton_utils.custom_cache_manager:CustomCacheManager
|
||||||
|
INFO 07-14 11:34:35 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
INFO 07-14 11:34:35 selector.py:115] Using XFormers backend.
|
||||||
|
/usr/local/corex/lib/python3/dist-packages/torch/cuda/__init__.py:51: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you.
|
||||||
|
import pynvml # type: ignore[import]
|
||||||
|
/usr/local/corex/lib/python3/dist-packages/torch/cuda/__init__.py:51: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you.
|
||||||
|
import pynvml # type: ignore[import]
|
||||||
|
/usr/local/corex/lib/python3/dist-packages/torch/cuda/__init__.py:51: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you.
|
||||||
|
import pynvml # type: ignore[import]
|
||||||
|
INFO 07-14 11:34:37 importing.py:10] Triton not installed; certain GPU-related functions will not be available.
|
||||||
|
INFO 07-14 11:34:37 importing.py:10] Triton not installed; certain GPU-related functions will not be available.
|
||||||
|
INFO 07-14 11:34:37 importing.py:10] Triton not installed; certain GPU-related functions will not be available.
|
||||||
|
WARNING:tensorflow:Deprecation warnings have been disabled. Set TF_ENABLE_DEPRECATION_WARNINGS=1 to re-enable them.
|
||||||
|
WARNING:tensorflow:Deprecation warnings have been disabled. Set TF_ENABLE_DEPRECATION_WARNINGS=1 to re-enable them.
|
||||||
|
WARNING:tensorflow:Deprecation warnings have been disabled. Set TF_ENABLE_DEPRECATION_WARNINGS=1 to re-enable them.
|
||||||
|
[1;36m(VllmWorkerProcess pid=10987)[0;0m INFO 07-14 11:34:44 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=10987)[0;0m INFO 07-14 11:34:44 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=10987)[0;0m INFO 07-14 11:34:44 multiproc_worker_utils.py:216] Worker ready; awaiting tasks
|
||||||
|
[1;36m(VllmWorkerProcess pid=10988)[0;0m INFO 07-14 11:34:44 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=10988)[0;0m INFO 07-14 11:34:44 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=10988)[0;0m INFO 07-14 11:34:44 multiproc_worker_utils.py:216] Worker ready; awaiting tasks
|
||||||
|
[1;36m(VllmWorkerProcess pid=10989)[0;0m INFO 07-14 11:34:44 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=10989)[0;0m INFO 07-14 11:34:44 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=10989)[0;0m INFO 07-14 11:34:44 multiproc_worker_utils.py:216] Worker ready; awaiting tasks
|
||||||
|
INFO 07-14 11:34:45 shm_broadcast.py:242] vLLM message queue communication handle: Handle(connect_ip='127.0.0.1', local_reader_ranks=[1, 2, 3], buffer=<vllm.distributed.device_communicators.shm_broadcast.ShmRingBuffer object at 0x7f5623116410>, local_subscribe_port=42265, remote_subscribe_port=None)
|
||||||
|
INFO 07-14 11:34:45 model_runner.py:1112] Starting to load model /root/public-storage/models/Qwen/Qwen3.6-35B-A3B...
|
||||||
|
[1;36m(VllmWorkerProcess pid=10987)[0;0m INFO 07-14 11:34:45 model_runner.py:1112] Starting to load model /root/public-storage/models/Qwen/Qwen3.6-35B-A3B...
|
||||||
|
[1;36m(VllmWorkerProcess pid=10988)[0;0m INFO 07-14 11:34:45 model_runner.py:1112] Starting to load model /root/public-storage/models/Qwen/Qwen3.6-35B-A3B...
|
||||||
|
[1;36m(VllmWorkerProcess pid=10989)[0;0m INFO 07-14 11:34:45 model_runner.py:1112] Starting to load model /root/public-storage/models/Qwen/Qwen3.6-35B-A3B...
|
||||||
|
INFO 07-14 11:34:45 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
INFO 07-14 11:34:45 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=10988)[0;0m INFO 07-14 11:34:45 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=10987)[0;0m INFO 07-14 11:34:45 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=10988)[0;0m INFO 07-14 11:34:45 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=10987)[0;0m INFO 07-14 11:34:45 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=10989)[0;0m INFO 07-14 11:34:45 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=10989)[0;0m INFO 07-14 11:34:45 selector.py:115] Using XFormers backend.
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 0% Completed | 0/26 [00:00<?, ?it/s]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 4% Completed | 1/26 [00:01<00:44, 1.79s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 8% Completed | 2/26 [00:02<00:23, 1.00it/s]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 12% Completed | 3/26 [00:04<00:31, 1.36s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 15% Completed | 4/26 [00:05<00:34, 1.57s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 19% Completed | 5/26 [00:06<00:28, 1.36s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 23% Completed | 6/26 [00:08<00:31, 1.58s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 27% Completed | 7/26 [00:09<00:22, 1.20s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 31% Completed | 8/26 [00:11<00:25, 1.39s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 35% Completed | 9/26 [00:12<00:21, 1.25s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 38% Completed | 10/26 [00:14<00:23, 1.47s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 42% Completed | 11/26 [00:14<00:16, 1.11s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 46% Completed | 12/26 [00:16<00:18, 1.33s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 50% Completed | 13/26 [00:18<00:19, 1.50s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 54% Completed | 14/26 [00:20<00:20, 1.68s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 58% Completed | 15/26 [00:20<00:15, 1.42s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 62% Completed | 16/26 [00:23<00:17, 1.72s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 65% Completed | 17/26 [00:23<00:11, 1.33s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 69% Completed | 18/26 [00:25<00:11, 1.47s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 73% Completed | 19/26 [00:27<00:10, 1.55s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 77% Completed | 20/26 [00:29<00:10, 1.83s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 81% Completed | 21/26 [00:30<00:07, 1.59s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 85% Completed | 22/26 [00:31<00:05, 1.33s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 88% Completed | 23/26 [00:33<00:04, 1.65s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 92% Completed | 24/26 [00:35<00:03, 1.72s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 96% Completed | 25/26 [00:37<00:01, 1.78s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 100% Completed | 26/26 [00:38<00:00, 1.40s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 100% Completed | 26/26 [00:38<00:00, 1.47s/it]
|
||||||
|
|
||||||
|
[1;36m(VllmWorkerProcess pid=10987)[0;0m INFO 07-14 11:35:24 model_runner.py:1123] Loading model weights took 16.2303 GB
|
||||||
|
INFO 07-14 11:35:24 model_runner.py:1123] Loading model weights took 16.2303 GB
|
||||||
|
[1;36m(VllmWorkerProcess pid=10989)[0;0m INFO 07-14 11:35:24 model_runner.py:1123] Loading model weights took 16.2303 GB
|
||||||
|
[1;36m(VllmWorkerProcess pid=10988)[0;0m INFO 07-14 11:35:24 model_runner.py:1123] Loading model weights took 16.2303 GB
|
||||||
|
INFO 07-14 11:35:33 distributed_gpu_executor.py:57] # GPU blocks: 21100, # CPU blocks: 6553
|
||||||
|
INFO 07-14 11:35:33 distributed_gpu_executor.py:61] Maximum concurrency for 100000 tokens per request: 3.38x
|
||||||
|
INFO 07-14 11:35:38 serving_chat.py:79] "auto" tool choice has been enabled please note that while the parallel_tool_calls client option is preset for compatibility reasons, it will be ignored.
|
||||||
|
INFO 07-14 11:35:38 serving_chat.py:101] Reasoning parser 'qwen3' enabled.
|
||||||
|
WARNING 07-14 11:35:38 serving_embedding.py:199] embedding_mode is False. Embedding API will not work.
|
||||||
|
INFO 07-14 11:35:38 launcher.py:19] Available routes are:
|
||||||
|
INFO 07-14 11:35:38 launcher.py:27] Route: /openapi.json, Methods: HEAD, GET
|
||||||
|
INFO 07-14 11:35:38 launcher.py:27] Route: /docs, Methods: HEAD, GET
|
||||||
|
INFO 07-14 11:35:38 launcher.py:27] Route: /docs/oauth2-redirect, Methods: HEAD, GET
|
||||||
|
INFO 07-14 11:35:38 launcher.py:27] Route: /redoc, Methods: HEAD, GET
|
||||||
|
INFO 07-14 11:35:38 launcher.py:27] Route: /health, Methods: GET
|
||||||
|
INFO 07-14 11:35:38 launcher.py:27] Route: /tokenize, Methods: POST
|
||||||
|
INFO 07-14 11:35:38 launcher.py:27] Route: /detokenize, Methods: POST
|
||||||
|
INFO 07-14 11:35:38 launcher.py:27] Route: /v1/models, Methods: GET
|
||||||
|
INFO 07-14 11:35:38 launcher.py:27] Route: /version, Methods: GET
|
||||||
|
INFO 07-14 11:35:38 launcher.py:27] Route: /v1/chat/completions, Methods: POST
|
||||||
|
INFO 07-14 11:35:38 launcher.py:27] Route: /v1/completions, Methods: POST
|
||||||
|
INFO 07-14 11:35:38 launcher.py:27] Route: /v1/embeddings, Methods: POST
|
||||||
|
INFO: Started server process [10648]
|
||||||
|
INFO: Waiting for application startup.
|
||||||
|
INFO: Application startup complete.
|
||||||
|
INFO: Uvicorn running on socket ('0.0.0.0', 1111) (Press CTRL+C to quit)
|
||||||
|
INFO 07-14 11:35:48 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:35:48 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:35:58 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:35:58 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:36:08 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:36:08 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:36:18 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:36:18 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:36:28 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:36:28 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:36:38 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:36:38 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:36:48 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:36:48 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO: 127.0.0.1:35518 - "GET /health HTTP/1.1" 200 OK
|
||||||
|
INFO 07-14 11:36:58 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:36:58 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:37:08 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:37:08 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:37:18 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:37:18 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:37:28 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:37:28 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:37:38 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:37:38 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:37:48 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:37:48 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:37:58 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:37:58 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:38:08 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:38:08 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO: 127.0.0.1:46034 - "GET /health HTTP/1.1" 200 OK
|
||||||
|
INFO 07-14 11:38:18 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:38:18 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:38:28 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:38:28 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:38:38 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:38:38 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:38:48 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:38:48 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO: 127.0.0.1:56962 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||||
|
/usr/local/lib/python3.10/site-packages/pyairports/airports.py:1: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81.
|
||||||
|
from pkg_resources import resource_string
|
||||||
|
INFO 07-14 11:38:58 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 1 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:38:58 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:39:01 qwen3_5.py:95] [ENGINEX_PROFILE_QWEN] rank=0 steps=16 mode=decode layer.mlp: total=6479.37ms avg=10.124ms n=640 | moe.routed_total: total=5732.82ms avg=8.958ms n=640 | moe.routed_prefill_experts: total=5138.29ms avg=64.229ms n=80 | layer.linear_attention: total=4877.66ms avg=10.162ms n=480 | layer.full_attention: total=701.53ms avg=4.385ms n=160 | full_attn.paged_attention: total=428.46ms avg=2.678ms n=160 | moe.tp_all_reduce: total=380.35ms avg=0.594ms n=640 | moe.routed_decode_experts: total=342.04ms avg=0.611ms n=560 | layer.input_norm: total=229.22ms avg=0.358ms n=640 | layer.post_attn_norm: total=216.38ms avg=0.338ms n=640 | moe.routing_topk: total=216.08ms avg=0.338ms n=640 | moe.shared_expert: total=209.64ms avg=0.328ms n=640 | full_attn.gate_o_proj: total=110.16ms avg=0.688ms n=160 | full_attn.norm_rope: total=84.41ms avg=0.528ms n=160 | full_attn.qkv_proj: total=63.63ms avg=0.398ms n=160 | moe.gate: total=57.67ms avg=0.090ms n=640 | moe.combine: total=32.51ms avg=0.051ms n=640
|
||||||
|
[1;36m(VllmWorkerProcess pid=10988)[0;0m INFO 07-14 11:39:01 qwen3_5.py:95] [ENGINEX_PROFILE_QWEN] rank=2 steps=16 mode=decode layer.mlp: total=6477.48ms avg=10.121ms n=640 | moe.routed_total: total=5756.75ms avg=8.995ms n=640 | moe.routed_prefill_experts: total=5166.53ms avg=64.582ms n=80 | layer.linear_attention: total=4881.80ms avg=10.170ms n=480 | layer.full_attention: total=701.10ms avg=4.382ms n=160 | full_attn.paged_attention: total=420.51ms avg=2.628ms n=160 | moe.routed_decode_experts: total=344.50ms avg=0.615ms n=560 | moe.tp_all_reduce: total=342.37ms avg=0.535ms n=640 | layer.input_norm: total=225.86ms avg=0.353ms n=640 | layer.post_attn_norm: total=218.22ms avg=0.341ms n=640 | moe.shared_expert: total=210.19ms avg=0.328ms n=640 | moe.routing_topk: total=209.87ms avg=0.328ms n=640 | full_attn.gate_o_proj: total=117.54ms avg=0.735ms n=160 | full_attn.norm_rope: total=84.89ms avg=0.531ms n=160 | full_attn.qkv_proj: total=63.45ms avg=0.397ms n=160 | moe.gate: total=59.30ms avg=0.093ms n=640 | moe.combine: total=42.91ms avg=0.067ms n=640
|
||||||
|
[1;36m(VllmWorkerProcess pid=10989)[0;0m INFO 07-14 11:39:01 qwen3_5.py:95] [ENGINEX_PROFILE_QWEN] rank=3 steps=16 mode=decode layer.mlp: total=6477.64ms avg=10.121ms n=640 | moe.routed_total: total=5780.92ms avg=9.033ms n=640 | moe.routed_prefill_experts: total=5196.83ms avg=64.960ms n=80 | layer.linear_attention: total=4881.38ms avg=10.170ms n=480 | layer.full_attention: total=701.03ms avg=4.381ms n=160 | full_attn.paged_attention: total=420.14ms avg=2.626ms n=160 | moe.routed_decode_experts: total=343.09ms avg=0.613ms n=560 | moe.tp_all_reduce: total=326.79ms avg=0.511ms n=640 | layer.input_norm: total=226.41ms avg=0.354ms n=640 | layer.post_attn_norm: total=218.14ms avg=0.341ms n=640 | moe.shared_expert: total=209.91ms avg=0.328ms n=640 | moe.routing_topk: total=204.80ms avg=0.320ms n=640 | full_attn.gate_o_proj: total=116.90ms avg=0.731ms n=160 | full_attn.norm_rope: total=85.65ms avg=0.535ms n=160 | full_attn.qkv_proj: total=63.79ms avg=0.399ms n=160 | moe.gate: total=59.58ms avg=0.093ms n=640 | moe.combine: total=34.85ms avg=0.054ms n=640
|
||||||
|
[1;36m(VllmWorkerProcess pid=10987)[0;0m INFO 07-14 11:39:01 qwen3_5.py:95] [ENGINEX_PROFILE_QWEN] rank=1 steps=16 mode=decode layer.mlp: total=6479.94ms avg=10.125ms n=640 | moe.routed_total: total=5735.46ms avg=8.962ms n=640 | moe.routed_prefill_experts: total=5154.84ms avg=64.436ms n=80 | layer.linear_attention: total=4881.74ms avg=10.170ms n=480 | layer.full_attention: total=702.39ms avg=4.390ms n=160 | full_attn.paged_attention: total=427.34ms avg=2.671ms n=160 | moe.tp_all_reduce: total=377.53ms avg=0.590ms n=640 | moe.routed_decode_experts: total=341.55ms avg=0.610ms n=560 | layer.input_norm: total=225.32ms avg=0.352ms n=640 | layer.post_attn_norm: total=215.76ms avg=0.337ms n=640 | moe.shared_expert: total=209.99ms avg=0.328ms n=640 | moe.routing_topk: total=203.00ms avg=0.317ms n=640 | full_attn.gate_o_proj: total=113.29ms avg=0.708ms n=160 | full_attn.norm_rope: total=83.79ms avg=0.524ms n=160 | full_attn.qkv_proj: total=63.15ms avg=0.395ms n=160 | moe.gate: total=57.93ms avg=0.091ms n=640 | moe.combine: total=33.78ms avg=0.053ms n=640
|
||||||
|
INFO 07-14 11:39:01 model_runner.py:114] [ENGINEX_PROFILE_MODEL_RUNNER] steps=16 mode=decode prompt.model_forward: total=10698.55ms avg=5349.273ms n=2 | decode.model_forward: total=2037.07ms avg=145.505ms n=14 | prompt.compute_logits: total=202.86ms avg=101.429ms n=2 | prompt.sample: total=40.62ms avg=20.308ms n=2 | decode.compute_logits: total=16.65ms avg=1.189ms n=14 | decode.sample: total=14.14ms avg=1.010ms n=14 | decode.attn_begin_forward: total=0.22ms avg=0.016ms n=14 | prompt.attn_begin_forward: total=0.05ms avg=0.025ms n=2
|
||||||
|
INFO 07-14 11:39:03 metrics.py:345] Avg prompt throughput: 7.7 tokens/s, Avg generation throughput: 4.8 tokens/s, Running: 1 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:39:03 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
[1;36m(VllmWorkerProcess pid=10988)[0;0m INFO 07-14 11:39:04 qwen3_5.py:95] [ENGINEX_PROFILE_QWEN] rank=2 steps=32 mode=decode layer.mlp: total=7512.25ms avg=5.869ms n=1280 | moe.routed_total: total=6320.62ms avg=4.938ms n=1280 | layer.linear_attention: total=5549.77ms avg=5.781ms n=960 | moe.routed_prefill_experts: total=5166.53ms avg=64.582ms n=80 | layer.full_attention: total=923.27ms avg=2.885ms n=320 | moe.routed_decode_experts: total=735.47ms avg=0.613ms n=1200 | moe.tp_all_reduce: total=531.76ms avg=0.415ms n=1280 | full_attn.paged_attention: total=463.63ms avg=1.449ms n=320 | moe.shared_expert: total=359.85ms avg=0.281ms n=1280 | layer.input_norm: total=357.80ms avg=0.280ms n=1280 | layer.post_attn_norm: total=349.09ms avg=0.273ms n=1280 | moe.routing_topk: total=348.27ms avg=0.272ms n=1280 | full_attn.gate_o_proj: total=184.63ms avg=0.577ms n=320 | full_attn.norm_rope: total=148.94ms avg=0.465ms n=320 | moe.gate: total=102.29ms avg=0.080ms n=1280 | full_attn.qkv_proj: total=97.08ms avg=0.303ms n=320 | moe.combine: total=67.92ms avg=0.053ms n=1280
|
||||||
|
INFO 07-14 11:39:04 qwen3_5.py:95] [ENGINEX_PROFILE_QWEN] rank=0 steps=32 mode=decode layer.mlp: total=7514.73ms avg=5.871ms n=1280 | moe.routed_total: total=6297.44ms avg=4.920ms n=1280 | layer.linear_attention: total=5544.35ms avg=5.775ms n=960 | moe.routed_prefill_experts: total=5138.29ms avg=64.229ms n=80 | layer.full_attention: total=923.60ms avg=2.886ms n=320 | moe.routed_decode_experts: total=731.56ms avg=0.610ms n=1200 | moe.tp_all_reduce: total=568.78ms avg=0.444ms n=1280 | full_attn.paged_attention: total=471.55ms avg=1.474ms n=320 | layer.input_norm: total=361.94ms avg=0.283ms n=1280 | moe.shared_expert: total=359.86ms avg=0.281ms n=1280 | moe.routing_topk: total=356.01ms avg=0.278ms n=1280 | layer.post_attn_norm: total=346.28ms avg=0.271ms n=1280 | full_attn.gate_o_proj: total=176.48ms avg=0.552ms n=320 | full_attn.norm_rope: total=148.59ms avg=0.464ms n=320 | moe.gate: total=100.46ms avg=0.078ms n=1280 | full_attn.qkv_proj: total=97.63ms avg=0.305ms n=320 | moe.combine: total=56.61ms avg=0.044ms n=1280
|
||||||
|
[1;36m(VllmWorkerProcess pid=10989)[0;0m INFO 07-14 11:39:04 qwen3_5.py:95] [ENGINEX_PROFILE_QWEN] rank=3 steps=32 mode=decode layer.mlp: total=7511.89ms avg=5.869ms n=1280 | moe.routed_total: total=6346.78ms avg=4.958ms n=1280 | layer.linear_attention: total=5546.91ms avg=5.778ms n=960 | moe.routed_prefill_experts: total=5196.83ms avg=64.960ms n=80 | layer.full_attention: total=923.09ms avg=2.885ms n=320 | moe.routed_decode_experts: total=734.27ms avg=0.612ms n=1200 | moe.tp_all_reduce: total=507.39ms avg=0.396ms n=1280 | full_attn.paged_attention: total=464.27ms avg=1.451ms n=320 | moe.shared_expert: total=361.08ms avg=0.282ms n=1280 | layer.input_norm: total=360.13ms avg=0.281ms n=1280 | layer.post_attn_norm: total=350.27ms avg=0.274ms n=1280 | moe.routing_topk: total=344.92ms avg=0.269ms n=1280 | full_attn.gate_o_proj: total=181.82ms avg=0.568ms n=320 | full_attn.norm_rope: total=150.20ms avg=0.469ms n=320 | moe.gate: total=104.89ms avg=0.082ms n=1280 | full_attn.qkv_proj: total=98.08ms avg=0.306ms n=320 | moe.combine: total=62.51ms avg=0.049ms n=1280
|
||||||
|
[1;36m(VllmWorkerProcess pid=10987)[0;0m INFO 07-14 11:39:04 qwen3_5.py:95] [ENGINEX_PROFILE_QWEN] rank=1 steps=32 mode=decode layer.mlp: total=7515.10ms avg=5.871ms n=1280 | moe.routed_total: total=6298.23ms avg=4.920ms n=1280 | layer.linear_attention: total=5549.96ms avg=5.781ms n=960 | moe.routed_prefill_experts: total=5154.84ms avg=64.436ms n=80 | layer.full_attention: total=924.95ms avg=2.890ms n=320 | moe.routed_decode_experts: total=731.88ms avg=0.610ms n=1200 | moe.tp_all_reduce: total=567.74ms avg=0.444ms n=1280 | full_attn.paged_attention: total=469.88ms avg=1.468ms n=320 | moe.shared_expert: total=359.93ms avg=0.281ms n=1280 | layer.input_norm: total=356.63ms avg=0.279ms n=1280 | layer.post_attn_norm: total=346.43ms avg=0.271ms n=1280 | moe.routing_topk: total=340.77ms avg=0.266ms n=1280 | full_attn.gate_o_proj: total=181.26ms avg=0.566ms n=320 | full_attn.norm_rope: total=147.83ms avg=0.462ms n=320 | moe.gate: total=101.03ms avg=0.079ms n=1280 | full_attn.qkv_proj: total=96.90ms avg=0.303ms n=320 | moe.combine: total=58.88ms avg=0.046ms n=1280
|
||||||
|
INFO 07-14 11:39:04 model_runner.py:114] [ENGINEX_PROFILE_MODEL_RUNNER] steps=32 mode=decode prompt.model_forward: total=10698.55ms avg=5349.273ms n=2 | decode.model_forward: total=4300.50ms avg=143.350ms n=30 | prompt.compute_logits: total=202.86ms avg=101.429ms n=2 | prompt.sample: total=40.62ms avg=20.308ms n=2 | decode.compute_logits: total=35.17ms avg=1.172ms n=30 | decode.sample: total=30.18ms avg=1.006ms n=30 | decode.attn_begin_forward: total=0.46ms avg=0.015ms n=30 | prompt.attn_begin_forward: total=0.05ms avg=0.025ms n=2
|
||||||
|
[1;36m(VllmWorkerProcess pid=10988)[0;0m INFO 07-14 11:39:06 qwen3_5.py:95] [ENGINEX_PROFILE_QWEN] rank=2 steps=48 mode=decode layer.mlp: total=8565.64ms avg=4.461ms n=1920 | moe.routed_total: total=6883.91ms avg=3.585ms n=1920 | layer.linear_attention: total=6230.86ms avg=4.327ms n=1440 | moe.routed_prefill_experts: total=5166.53ms avg=64.582ms n=80 | layer.full_attention: total=1149.78ms avg=2.395ms n=480 | moe.routed_decode_experts: total=1125.25ms avg=0.612ms n=1840 | moe.tp_all_reduce: total=740.06ms avg=0.385ms n=1920 | moe.shared_expert: total=510.07ms avg=0.266ms n=1920 | full_attn.paged_attention: total=507.77ms avg=1.058ms n=480 | layer.input_norm: total=490.33ms avg=0.255ms n=1920 | moe.routing_topk: total=487.27ms avg=0.254ms n=1920 | layer.post_attn_norm: total=478.23ms avg=0.249ms n=1920 | full_attn.gate_o_proj: total=254.73ms avg=0.531ms n=480 | full_attn.norm_rope: total=213.20ms avg=0.444ms n=480 | moe.gate: total=144.81ms avg=0.075ms n=1920 | full_attn.qkv_proj: total=130.70ms avg=0.272ms n=480 | moe.combine: total=92.98ms avg=0.048ms n=1920
|
||||||
|
INFO 07-14 11:39:06 qwen3_5.py:95] [ENGINEX_PROFILE_QWEN] rank=0 steps=48 mode=decode layer.mlp: total=8565.83ms avg=4.461ms n=1920 | moe.routed_total: total=6862.20ms avg=3.574ms n=1920 | layer.linear_attention: total=6220.48ms avg=4.320ms n=1440 | moe.routed_prefill_experts: total=5138.29ms avg=64.229ms n=80 | layer.full_attention: total=1149.99ms avg=2.396ms n=480 | moe.routed_decode_experts: total=1121.57ms avg=0.610ms n=1840 | moe.tp_all_reduce: total=770.33ms avg=0.401ms n=1920 | full_attn.paged_attention: total=515.91ms avg=1.075ms n=480 | moe.shared_expert: total=511.62ms avg=0.266ms n=1920 | layer.input_norm: total=497.87ms avg=0.259ms n=1920 | moe.routing_topk: total=495.48ms avg=0.258ms n=1920 | layer.post_attn_norm: total=476.22ms avg=0.248ms n=1920 | full_attn.gate_o_proj: total=245.20ms avg=0.511ms n=480 | full_attn.norm_rope: total=213.11ms avg=0.444ms n=480 | moe.gate: total=143.33ms avg=0.075ms n=1920 | full_attn.qkv_proj: total=131.71ms avg=0.274ms n=480 | moe.combine: total=80.87ms avg=0.042ms n=1920
|
||||||
|
[1;36m(VllmWorkerProcess pid=10987)[0;0m INFO 07-14 11:39:06 qwen3_5.py:95] [ENGINEX_PROFILE_QWEN] rank=1 steps=48 mode=decode layer.mlp: total=8565.70ms avg=4.461ms n=1920 | moe.routed_total: total=6865.80ms avg=3.576ms n=1920 | layer.linear_attention: total=6230.90ms avg=4.327ms n=1440 | moe.routed_prefill_experts: total=5154.84ms avg=64.436ms n=80 | layer.full_attention: total=1151.16ms avg=2.398ms n=480 | moe.routed_decode_experts: total=1124.60ms avg=0.611ms n=1840 | moe.tp_all_reduce: total=764.42ms avg=0.398ms n=1920 | full_attn.paged_attention: total=513.29ms avg=1.069ms n=480 | moe.shared_expert: total=512.39ms avg=0.267ms n=1920 | layer.input_norm: total=490.33ms avg=0.255ms n=1920 | moe.routing_topk: total=480.44ms avg=0.250ms n=1920 | layer.post_attn_norm: total=477.60ms avg=0.249ms n=1920 | full_attn.gate_o_proj: total=249.26ms avg=0.519ms n=480 | full_attn.norm_rope: total=213.91ms avg=0.446ms n=480 | moe.gate: total=144.66ms avg=0.075ms n=1920 | full_attn.qkv_proj: total=131.07ms avg=0.273ms n=480 | moe.combine: total=84.43ms avg=0.044ms n=1920
|
||||||
|
[1;36m(VllmWorkerProcess pid=10989)[0;0m INFO 07-14 11:39:06 qwen3_5.py:95] [ENGINEX_PROFILE_QWEN] rank=3 steps=48 mode=decode layer.mlp: total=8560.84ms avg=4.459ms n=1920 | moe.routed_total: total=6913.41ms avg=3.601ms n=1920 | layer.linear_attention: total=6224.60ms avg=4.323ms n=1440 | moe.routed_prefill_experts: total=5196.83ms avg=64.960ms n=80 | layer.full_attention: total=1149.35ms avg=2.394ms n=480 | moe.routed_decode_experts: total=1124.91ms avg=0.611ms n=1840 | moe.tp_all_reduce: total=700.04ms avg=0.365ms n=1920 | moe.shared_expert: total=513.69ms avg=0.268ms n=1920 | full_attn.paged_attention: total=509.40ms avg=1.061ms n=480 | layer.input_norm: total=497.02ms avg=0.259ms n=1920 | moe.routing_topk: total=486.26ms avg=0.253ms n=1920 | layer.post_attn_norm: total=482.80ms avg=0.251ms n=1920 | full_attn.gate_o_proj: total=249.50ms avg=0.520ms n=480 | full_attn.norm_rope: total=215.13ms avg=0.448ms n=480 | moe.gate: total=150.55ms avg=0.078ms n=1920 | full_attn.qkv_proj: total=132.46ms avg=0.276ms n=480 | moe.combine: total=90.08ms avg=0.047ms n=1920
|
||||||
|
INFO 07-14 11:39:06 model_runner.py:114] [ENGINEX_PROFILE_MODEL_RUNNER] steps=48 mode=decode prompt.model_forward: total=10698.55ms avg=5349.273ms n=2 | decode.model_forward: total=6600.11ms avg=143.481ms n=46 | prompt.compute_logits: total=202.86ms avg=101.429ms n=2 | decode.compute_logits: total=54.04ms avg=1.175ms n=46 | decode.sample: total=46.30ms avg=1.007ms n=46 | prompt.sample: total=40.62ms avg=20.308ms n=2 | decode.attn_begin_forward: total=0.71ms avg=0.015ms n=46 | prompt.attn_begin_forward: total=0.05ms avg=0.025ms n=2
|
||||||
|
INFO: 127.0.0.1:44782 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||||
|
INFO: 127.0.0.1:44794 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||||
|
INFO 07-14 11:39:08 metrics.py:345] Avg prompt throughput: 14.9 tokens/s, Avg generation throughput: 5.0 tokens/s, Running: 2 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:39:08 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
[1;36m(VllmWorkerProcess pid=10988)[0;0m INFO 07-14 11:39:12 qwen3_5.py:95] [ENGINEX_PROFILE_QWEN] rank=2 steps=64 mode=decode layer.mlp: total=11760.09ms avg=4.594ms n=2560 | moe.routed_total: total=9359.14ms avg=3.656ms n=2560 | moe.routed_prefill_experts: total=7427.27ms avg=10.922ms n=680 | layer.linear_attention: total=7360.67ms avg=3.834ms n=1920 | layer.full_attention: total=1439.20ms avg=2.249ms n=640 | moe.routed_decode_experts: total=1149.53ms avg=0.611ms n=1880 | moe.tp_all_reduce: total=1129.49ms avg=0.441ms n=2560 | moe.shared_expert: total=683.54ms avg=0.267ms n=2560 | layer.input_norm: total=653.91ms avg=0.255ms n=2560 | moe.routing_topk: total=632.62ms avg=0.247ms n=2560 | layer.post_attn_norm: total=630.36ms avg=0.246ms n=2560 | full_attn.paged_attention: total=570.22ms avg=0.891ms n=640 | full_attn.gate_o_proj: total=349.73ms avg=0.546ms n=640 | full_attn.norm_rope: total=288.81ms avg=0.451ms n=640 | moe.gate: total=193.10ms avg=0.075ms n=2560 | full_attn.qkv_proj: total=169.83ms avg=0.265ms n=640 | moe.combine: total=121.20ms avg=0.047ms n=2560
|
||||||
|
[1;36m(VllmWorkerProcess pid=10989)[0;0m INFO 07-14 11:39:12 qwen3_5.py:95] [ENGINEX_PROFILE_QWEN] rank=3 steps=64 mode=decode layer.mlp: total=11752.09ms avg=4.591ms n=2560 | moe.routed_total: total=9479.63ms avg=3.703ms n=2560 | moe.routed_prefill_experts: total=7543.04ms avg=11.093ms n=680 | layer.linear_attention: total=7352.66ms avg=3.830ms n=1920 | layer.full_attention: total=1437.73ms avg=2.246ms n=640 | moe.routed_decode_experts: total=1149.26ms avg=0.611ms n=1880 | moe.tp_all_reduce: total=987.66ms avg=0.386ms n=2560 | moe.shared_expert: total=690.96ms avg=0.270ms n=2560 | layer.input_norm: total=663.22ms avg=0.259ms n=2560 | layer.post_attn_norm: total=638.01ms avg=0.249ms n=2560 | moe.routing_topk: total=636.55ms avg=0.249ms n=2560 | full_attn.paged_attention: total=574.24ms avg=0.897ms n=640 | full_attn.gate_o_proj: total=337.69ms avg=0.528ms n=640 | full_attn.norm_rope: total=293.83ms avg=0.459ms n=640 | moe.gate: total=201.22ms avg=0.079ms n=2560 | full_attn.qkv_proj: total=171.31ms avg=0.268ms n=640 | moe.combine: total=120.35ms avg=0.047ms n=2560
|
||||||
|
INFO 07-14 11:39:12 qwen3_5.py:95] [ENGINEX_PROFILE_QWEN] rank=0 steps=64 mode=decode layer.mlp: total=11758.85ms avg=4.593ms n=2560 | moe.routed_total: total=9348.77ms avg=3.652ms n=2560 | moe.routed_prefill_experts: total=7406.06ms avg=10.891ms n=680 | layer.linear_attention: total=7349.52ms avg=3.828ms n=1920 | layer.full_attention: total=1438.51ms avg=2.248ms n=640 | moe.tp_all_reduce: total=1146.09ms avg=0.448ms n=2560 | moe.routed_decode_experts: total=1145.85ms avg=0.609ms n=1880 | moe.shared_expert: total=685.01ms avg=0.268ms n=2560 | layer.input_norm: total=662.05ms avg=0.259ms n=2560 | moe.routing_topk: total=643.44ms avg=0.251ms n=2560 | layer.post_attn_norm: total=628.23ms avg=0.245ms n=2560 | full_attn.paged_attention: total=578.66ms avg=0.904ms n=640 | full_attn.gate_o_proj: total=334.92ms avg=0.523ms n=640 | full_attn.norm_rope: total=291.31ms avg=0.455ms n=640 | moe.gate: total=191.05ms avg=0.075ms n=2560 | full_attn.qkv_proj: total=171.27ms avg=0.268ms n=640 | moe.combine: total=108.34ms avg=0.042ms n=2560
|
||||||
|
[1;36m(VllmWorkerProcess pid=10987)[0;0m INFO 07-14 11:39:12 qwen3_5.py:95] [ENGINEX_PROFILE_QWEN] rank=1 steps=64 mode=decode layer.mlp: total=11758.80ms avg=4.593ms n=2560 | moe.routed_total: total=9366.70ms avg=3.659ms n=2560 | moe.routed_prefill_experts: total=7438.39ms avg=10.939ms n=680 | layer.linear_attention: total=7358.41ms avg=3.833ms n=1920 | layer.full_attention: total=1440.15ms avg=2.250ms n=640 | moe.routed_decode_experts: total=1148.88ms avg=0.611ms n=1880 | moe.tp_all_reduce: total=1118.88ms avg=0.437ms n=2560 | moe.shared_expert: total=690.07ms avg=0.270ms n=2560 | layer.input_norm: total=656.06ms avg=0.256ms n=2560 | layer.post_attn_norm: total=629.74ms avg=0.246ms n=2560 | moe.routing_topk: total=626.92ms avg=0.245ms n=2560 | full_attn.paged_attention: total=576.96ms avg=0.901ms n=640 | full_attn.gate_o_proj: total=338.44ms avg=0.529ms n=640 | full_attn.norm_rope: total=291.45ms avg=0.455ms n=640 | moe.gate: total=193.07ms avg=0.075ms n=2560 | full_attn.qkv_proj: total=171.28ms avg=0.268ms n=640 | moe.combine: total=113.70ms avg=0.044ms n=2560
|
||||||
|
INFO 07-14 11:39:12 model_runner.py:114] [ENGINEX_PROFILE_MODEL_RUNNER] steps=64 mode=decode prompt.model_forward: total=11655.73ms avg=3885.243ms n=3 | decode.model_forward: total=10665.29ms avg=174.841ms n=61 | prompt.compute_logits: total=204.78ms avg=68.261ms n=3 | decode.compute_logits: total=74.24ms avg=1.217ms n=61 | decode.sample: total=63.94ms avg=1.048ms n=61 | prompt.sample: total=42.29ms avg=14.097ms n=3 | decode.attn_begin_forward: total=1.00ms avg=0.016ms n=61 | prompt.attn_begin_forward: total=0.07ms avg=0.023ms n=3
|
||||||
|
INFO 07-14 11:39:13 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 7.1 tokens/s, Running: 2 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:39:13 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
[1;36m(VllmWorkerProcess pid=10988)[0;0m INFO 07-14 11:39:16 qwen3_5.py:95] [ENGINEX_PROFILE_QWEN] rank=2 steps=80 mode=decode layer.mlp: total=14470.51ms avg=4.522ms n=3200 | moe.routed_total: total=11510.65ms avg=3.597ms n=3200 | moe.routed_prefill_experts: total=9399.98ms avg=7.121ms n=1320 | layer.linear_attention: total=8056.26ms avg=3.357ms n=2400 | layer.full_attention: total=1673.93ms avg=2.092ms n=800 | moe.tp_all_reduce: total=1391.61ms avg=0.435ms n=3200 | moe.routed_decode_experts: total=1149.53ms avg=0.611ms n=1880 | moe.shared_expert: total=842.90ms avg=0.263ms n=3200 | layer.input_norm: total=807.29ms avg=0.252ms n=3200 | moe.routing_topk: total=773.90ms avg=0.242ms n=3200 | layer.post_attn_norm: total=772.95ms avg=0.242ms n=3200 | full_attn.paged_attention: total=614.17ms avg=0.768ms n=800 | full_attn.gate_o_proj: total=418.93ms avg=0.524ms n=800 | full_attn.norm_rope: total=359.94ms avg=0.450ms n=800 | moe.gate: total=237.22ms avg=0.074ms n=3200 | full_attn.qkv_proj: total=205.89ms avg=0.257ms n=800 | moe.combine: total=148.40ms avg=0.046ms n=3200
|
||||||
|
[1;36m(VllmWorkerProcess pid=10989)[0;0m INFO 07-14 11:39:16 qwen3_5.py:95] [ENGINEX_PROFILE_QWEN] rank=3 steps=80 mode=decode layer.mlp: total=14463.88ms avg=4.520ms n=3200 | moe.routed_total: total=11683.10ms avg=3.651ms n=3200 | moe.routed_prefill_experts: total=9566.20ms avg=7.247ms n=1320 | layer.linear_attention: total=8045.62ms avg=3.352ms n=2400 | layer.full_attention: total=1671.73ms avg=2.090ms n=800 | moe.tp_all_reduce: total=1195.25ms avg=0.374ms n=3200 | moe.routed_decode_experts: total=1149.26ms avg=0.611ms n=1880 | moe.shared_expert: total=852.38ms avg=0.266ms n=3200 | layer.input_norm: total=817.45ms avg=0.255ms n=3200 | layer.post_attn_norm: total=781.93ms avg=0.244ms n=3200 | moe.routing_topk: total=779.28ms avg=0.244ms n=3200 | full_attn.paged_attention: total=618.81ms avg=0.774ms n=800 | full_attn.gate_o_proj: total=406.43ms avg=0.508ms n=800 | full_attn.norm_rope: total=365.17ms avg=0.456ms n=800 | moe.gate: total=246.83ms avg=0.077ms n=3200 | full_attn.qkv_proj: total=206.17ms avg=0.258ms n=800 | moe.combine: total=148.03ms avg=0.046ms n=3200
|
||||||
|
INFO 07-14 11:39:16 qwen3_5.py:95] [ENGINEX_PROFILE_QWEN] rank=0 steps=80 mode=decode layer.mlp: total=14467.70ms avg=4.521ms n=3200 | moe.routed_total: total=11514.73ms avg=3.598ms n=3200 | moe.routed_prefill_experts: total=9391.98ms avg=7.115ms n=1320 | layer.linear_attention: total=8045.36ms avg=3.352ms n=2400 | layer.full_attention: total=1672.70ms avg=2.091ms n=800 | moe.tp_all_reduce: total=1390.05ms avg=0.434ms n=3200 | moe.routed_decode_experts: total=1145.85ms avg=0.609ms n=1880 | moe.shared_expert: total=845.93ms avg=0.264ms n=3200 | layer.input_norm: total=815.85ms avg=0.255ms n=3200 | moe.routing_topk: total=785.51ms avg=0.245ms n=3200 | layer.post_attn_norm: total=771.29ms avg=0.241ms n=3200 | full_attn.paged_attention: total=622.70ms avg=0.778ms n=800 | full_attn.gate_o_proj: total=403.63ms avg=0.505ms n=800 | full_attn.norm_rope: total=362.91ms avg=0.454ms n=800 | moe.gate: total=235.10ms avg=0.073ms n=3200 | full_attn.qkv_proj: total=206.57ms avg=0.258ms n=800 | moe.combine: total=134.88ms avg=0.042ms n=3200
|
||||||
|
[1;36m(VllmWorkerProcess pid=10987)[0;0m INFO 07-14 11:39:16 qwen3_5.py:95] [ENGINEX_PROFILE_QWEN] rank=1 steps=80 mode=decode layer.mlp: total=14469.28ms avg=4.522ms n=3200 | moe.routed_total: total=11523.73ms avg=3.601ms n=3200 | moe.routed_prefill_experts: total=9417.05ms avg=7.134ms n=1320 | layer.linear_attention: total=8054.17ms avg=3.356ms n=2400 | layer.full_attention: total=1674.95ms avg=2.094ms n=800 | moe.tp_all_reduce: total=1373.87ms avg=0.429ms n=3200 | moe.routed_decode_experts: total=1148.88ms avg=0.611ms n=1880 | moe.shared_expert: total=850.87ms avg=0.266ms n=3200 | layer.input_norm: total=809.03ms avg=0.253ms n=3200 | layer.post_attn_norm: total=772.33ms avg=0.241ms n=3200 | moe.routing_topk: total=767.56ms avg=0.240ms n=3200 | full_attn.paged_attention: total=620.61ms avg=0.776ms n=800 | full_attn.gate_o_proj: total=409.41ms avg=0.512ms n=800 | full_attn.norm_rope: total=362.21ms avg=0.453ms n=800 | moe.gate: total=237.50ms avg=0.074ms n=3200 | full_attn.qkv_proj: total=206.08ms avg=0.258ms n=800 | moe.combine: total=140.22ms avg=0.044ms n=3200
|
||||||
|
INFO 07-14 11:39:16 model_runner.py:114] [ENGINEX_PROFILE_MODEL_RUNNER] steps=80 mode=decode decode.model_forward: total=14680.40ms avg=190.655ms n=77 | prompt.model_forward: total=11655.73ms avg=3885.243ms n=3 | prompt.compute_logits: total=204.78ms avg=68.261ms n=3 | decode.compute_logits: total=93.92ms avg=1.220ms n=77 | decode.sample: total=81.77ms avg=1.062ms n=77 | prompt.sample: total=42.29ms avg=14.097ms n=3 | decode.attn_begin_forward: total=1.26ms avg=0.016ms n=77 | prompt.attn_begin_forward: total=0.07ms avg=0.023ms n=3
|
||||||
|
INFO 07-14 11:39:28 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 1.8 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:39:28 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:39:38 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:39:38 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:39:48 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:39:48 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:39:58 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:39:58 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:40:08 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:40:08 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:40:18 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:40:18 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:40:28 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:40:28 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:40:38 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:40:38 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:40:48 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:40:48 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:40:58 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:40:58 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:41:08 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:41:08 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:41:18 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:41:18 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:41:28 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:41:28 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:41:38 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:41:38 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:41:48 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:41:48 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:41:58 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:41:58 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:42:08 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:42:08 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:42:18 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:42:18 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:42:28 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:42:28 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:42:38 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:42:38 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:42:48 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:42:48 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:42:58 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:42:58 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:43:08 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:43:08 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:43:18 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:43:18 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:43:28 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:43:28 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:43:38 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:43:38 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:43:48 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:43:48 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:43:58 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:43:58 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:44:08 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:44:08 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:44:18 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:44:18 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:44:28 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:44:28 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:44:38 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:44:38 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:44:48 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:44:48 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:44:58 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:44:58 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:45:08 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:45:08 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:45:18 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:45:18 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:45:28 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:45:28 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:45:38 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:45:38 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:45:48 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:45:48 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:45:58 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:45:58 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:46:08 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:46:08 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:46:18 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:46:18 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:46:28 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:46:28 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:46:38 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:46:38 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:46:48 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:46:48 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:46:58 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:46:58 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:47:08 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:47:08 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:47:18 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:47:18 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:47:28 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:47:28 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:47:38 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:47:38 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:47:48 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:47:48 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:47:58 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:47:58 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:48:08 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:48:08 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:48:18 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:48:18 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:48:28 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:48:28 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:48:38 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:48:38 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:48:48 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:48:48 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:48:58 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:48:58 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:49:08 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:49:08 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:49:18 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:49:18 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:49:28 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:49:28 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:49:38 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:49:38 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:49:48 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:49:48 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:49:58 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:49:58 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:50:08 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:50:08 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:50:18 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:50:18 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:50:28 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:50:28 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:50:38 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:50:38 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:50:48 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:50:48 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:50:58 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:50:58 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:51:08 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:51:08 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:51:18 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:51:18 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:51:28 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:51:28 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:51:38 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:51:38 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:51:48 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:51:48 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:51:58 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:51:58 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:52:08 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:52:08 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:52:18 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:52:18 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:52:28 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:52:28 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:52:38 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:52:38 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:52:48 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:52:48 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:52:58 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:52:58 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:53:08 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:53:08 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:53:18 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:53:18 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:53:28 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:53:28 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:53:38 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:53:38 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:53:48 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:53:48 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:53:58 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:53:58 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:54:08 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:54:08 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:54:18 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:54:18 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:54:28 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:54:28 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:54:38 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:54:38 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:54:48 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:54:48 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:54:58 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:54:58 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:55:08 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:55:08 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:55:18 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:55:18 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:55:28 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:55:28 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
@@ -0,0 +1,325 @@
|
|||||||
|
/usr/local/corex/lib/python3/dist-packages/torch/cuda/__init__.py:51: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you.
|
||||||
|
import pynvml # type: ignore[import]
|
||||||
|
INFO 07-14 11:57:38 importing.py:10] Triton not installed; certain GPU-related functions will not be available.
|
||||||
|
2026-07-14 11:57:40.260421: I tensorflow/core/util/port.cc:110] oneDNN custom operations are on. You may see slightly different numerical results due to floating-point round-off errors from different computation orders. To turn them off, set the environment variable `TF_ENABLE_ONEDNN_OPTS=0`.
|
||||||
|
2026-07-14 11:57:40.312205: I tensorflow/core/platform/cpu_feature_guard.cc:182] This TensorFlow binary is optimized to use available CPU instructions in performance-critical operations.
|
||||||
|
To enable the following instructions: SSE3 SSE4.1 SSE4.2 AVX AVX2 AVX512F AVX512_VNNI AVX512_BF16 AVX_VNNI AMX_TILE AMX_INT8 AMX_BF16 FMA, in other operations, rebuild TensorFlow with the appropriate compiler flags.
|
||||||
|
WARNING:tensorflow:Deprecation warnings have been disabled. Set TF_ENABLE_DEPRECATION_WARNINGS=1 to re-enable them.
|
||||||
|
INFO 07-14 11:57:45 api_server.py:530] vLLM API server version 0.6.3
|
||||||
|
INFO 07-14 11:57:45 api_server.py:531] args: Namespace(host='0.0.0.0', port=1111, uvicorn_log_level='info', allow_credentials=False, allowed_origins=['*'], allowed_methods=['*'], allowed_headers=['*'], api_key=None, lora_modules=None, prompt_adapters=None, chat_template=None, response_role='assistant', ssl_keyfile=None, ssl_certfile=None, ssl_ca_certs=None, ssl_cert_reqs=0, root_path=None, middleware=[], return_tokens_as_token_ids=False, disable_frontend_multiprocessing=True, enable_auto_tool_choice=True, tool_call_parser='qwen3_coder', tool_parser_plugin='', reasoning_parser='qwen3', model='/root/public-storage/models/Qwen/Qwen3.6-35B-A3B', tokenizer=None, skip_tokenizer_init=False, revision=None, code_revision=None, tokenizer_revision=None, tokenizer_mode='auto', trust_remote_code=True, download_dir=None, load_format='auto', config_format='auto', dtype='auto', kv_cache_dtype='auto', quantization_param_path=None, max_model_len=100000, guided_decoding_backend='outlines', distributed_executor_backend=None, worker_use_ray=False, pipeline_parallel_size=1, tensor_parallel_size=4, max_parallel_loading_workers=None, ray_workers_use_nsight=False, block_size=16, enable_prefix_caching=True, disable_sliding_window=False, use_v2_block_manager=True, num_lookahead_slots=0, seed=0, swap_space=4, cpu_offload_gb=0, gpu_memory_utilization=0.95, num_gpu_blocks_override=None, max_num_batched_tokens=8192, max_num_seqs=2, max_logprobs=20, disable_log_stats=False, quantization=None, rope_scaling=None, rope_theta=None, enforce_eager=True, max_context_len_to_capture=None, max_seq_len_to_capture=32768, disable_custom_all_reduce=False, tokenizer_pool_size=0, tokenizer_pool_type='ray', tokenizer_pool_extra_config=None, limit_mm_per_prompt=None, mm_processor_kwargs=None, enable_lora=False, max_loras=1, max_lora_rank=16, lora_extra_vocab_size=256, lora_dtype='auto', long_lora_scaling_factors=None, max_cpu_loras=None, fully_sharded_loras=False, enable_prompt_adapter=False, max_prompt_adapters=1, max_prompt_adapter_token=0, device='auto', num_scheduler_steps=1, multi_step_stream_outputs=True, scheduler_delay_factor=0.0, enable_chunked_prefill=True, speculative_model=None, speculative_model_quantization=None, num_speculative_tokens=None, speculative_disable_mqa_scorer=False, speculative_draft_tensor_parallel_size=None, speculative_max_model_len=None, speculative_disable_by_batch_size=None, ngram_prompt_lookup_max=None, ngram_prompt_lookup_min=None, spec_decoding_acceptance_method='rejection_sampler', typical_acceptance_sampler_posterior_threshold=None, typical_acceptance_sampler_posterior_alpha=None, disable_logprobs_during_spec_decoding=None, model_loader_extra_config=None, ignore_patterns=[], preemption_mode=None, served_model_name=['llm'], qlora_adapter_name_or_path=None, otlp_traces_endpoint=None, collect_detailed_traces=None, disable_async_output_proc=False, override_neuron_config=None, scheduling_policy='fcfs', disable_log_requests=True, max_log_len=None, disable_fastapi_docs=False)
|
||||||
|
INFO 07-14 11:57:45 config.py:1670] Downcasting torch.float32 to torch.float16.
|
||||||
|
INFO 07-14 11:57:56 config.py:887] Defaulting to use mp for distributed inference
|
||||||
|
INFO 07-14 11:57:56 config.py:1005] Chunked prefill is enabled with max_num_batched_tokens=8192.
|
||||||
|
WARNING 07-14 11:57:56 config.py:380] To see benefits of async output processing, enable CUDA graph. Since, enforce-eager is enabled, async output processor cannot be used
|
||||||
|
INFO 07-14 11:57:56 llm_engine.py:237] Initializing an LLM engine (v0.6.3) with config: model='/root/public-storage/models/Qwen/Qwen3.6-35B-A3B', speculative_config=None, tokenizer='/root/public-storage/models/Qwen/Qwen3.6-35B-A3B', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, override_neuron_config=None, rope_scaling=None, rope_theta=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.float16, max_seq_len=100000, download_dir=None, load_format=LoadFormat.AUTO, tensor_parallel_size=4, pipeline_parallel_size=1, disable_custom_all_reduce=False, quantization=None, enforce_eager=True, kv_cache_dtype=auto, quantization_param_path=None, device_config=cuda, decoding_config=DecodingConfig(guided_decoding_backend='outlines'), observability_config=ObservabilityConfig(otlp_traces_endpoint=None, collect_model_forward_time=False, collect_model_execute_time=False), seed=0, served_model_name=llm, use_v2_block_manager=True, num_scheduler_steps=1, chunked_prefill_enabled=True multi_step_stream_outputs=True, enable_prefix_caching=True, use_async_output_proc=False, use_cached_outputs=False, mm_processor_kwargs=None)
|
||||||
|
WARNING 07-14 11:57:57 multiproc_gpu_executor.py:53] Reducing Torch parallelism from 64 threads to 1 to avoid unnecessary CPU contention. Set OMP_NUM_THREADS in the external environment to tune this value as needed.
|
||||||
|
INFO 07-14 11:57:57 custom_cache_manager.py:17] Setting Triton cache manager to: vllm.triton_utils.custom_cache_manager:CustomCacheManager
|
||||||
|
INFO 07-14 11:57:57 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
INFO 07-14 11:57:57 selector.py:115] Using XFormers backend.
|
||||||
|
/usr/local/corex/lib/python3/dist-packages/torch/cuda/__init__.py:51: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you.
|
||||||
|
import pynvml # type: ignore[import]
|
||||||
|
/usr/local/corex/lib/python3/dist-packages/torch/cuda/__init__.py:51: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you.
|
||||||
|
import pynvml # type: ignore[import]
|
||||||
|
/usr/local/corex/lib/python3/dist-packages/torch/cuda/__init__.py:51: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you.
|
||||||
|
import pynvml # type: ignore[import]
|
||||||
|
INFO 07-14 11:57:59 importing.py:10] Triton not installed; certain GPU-related functions will not be available.
|
||||||
|
INFO 07-14 11:57:59 importing.py:10] Triton not installed; certain GPU-related functions will not be available.
|
||||||
|
INFO 07-14 11:57:59 importing.py:10] Triton not installed; certain GPU-related functions will not be available.
|
||||||
|
WARNING:tensorflow:Deprecation warnings have been disabled. Set TF_ENABLE_DEPRECATION_WARNINGS=1 to re-enable them.
|
||||||
|
WARNING:tensorflow:Deprecation warnings have been disabled. Set TF_ENABLE_DEPRECATION_WARNINGS=1 to re-enable them.
|
||||||
|
WARNING:tensorflow:Deprecation warnings have been disabled. Set TF_ENABLE_DEPRECATION_WARNINGS=1 to re-enable them.
|
||||||
|
[1;36m(VllmWorkerProcess pid=12142)[0;0m INFO 07-14 11:58:06 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=12142)[0;0m INFO 07-14 11:58:06 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=12142)[0;0m INFO 07-14 11:58:06 multiproc_worker_utils.py:216] Worker ready; awaiting tasks
|
||||||
|
[1;36m(VllmWorkerProcess pid=12140)[0;0m INFO 07-14 11:58:06 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=12140)[0;0m INFO 07-14 11:58:06 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=12140)[0;0m INFO 07-14 11:58:06 multiproc_worker_utils.py:216] Worker ready; awaiting tasks
|
||||||
|
[1;36m(VllmWorkerProcess pid=12141)[0;0m INFO 07-14 11:58:06 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=12141)[0;0m INFO 07-14 11:58:06 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=12141)[0;0m INFO 07-14 11:58:06 multiproc_worker_utils.py:216] Worker ready; awaiting tasks
|
||||||
|
INFO 07-14 11:58:07 shm_broadcast.py:242] vLLM message queue communication handle: Handle(connect_ip='127.0.0.1', local_reader_ranks=[1, 2, 3], buffer=<vllm.distributed.device_communicators.shm_broadcast.ShmRingBuffer object at 0x7f815230ec50>, local_subscribe_port=60007, remote_subscribe_port=None)
|
||||||
|
INFO 07-14 11:58:07 model_runner.py:1112] Starting to load model /root/public-storage/models/Qwen/Qwen3.6-35B-A3B...
|
||||||
|
[1;36m(VllmWorkerProcess pid=12140)[0;0m INFO 07-14 11:58:07 model_runner.py:1112] Starting to load model /root/public-storage/models/Qwen/Qwen3.6-35B-A3B...
|
||||||
|
[1;36m(VllmWorkerProcess pid=12141)[0;0m INFO 07-14 11:58:07 model_runner.py:1112] Starting to load model /root/public-storage/models/Qwen/Qwen3.6-35B-A3B...
|
||||||
|
[1;36m(VllmWorkerProcess pid=12142)[0;0m INFO 07-14 11:58:07 model_runner.py:1112] Starting to load model /root/public-storage/models/Qwen/Qwen3.6-35B-A3B...
|
||||||
|
INFO 07-14 11:58:07 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
INFO 07-14 11:58:07 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=12142)[0;0m INFO 07-14 11:58:07 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=12140)[0;0m INFO 07-14 11:58:07 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=12141)[0;0m INFO 07-14 11:58:07 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=12142)[0;0m INFO 07-14 11:58:07 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=12140)[0;0m INFO 07-14 11:58:07 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=12141)[0;0m INFO 07-14 11:58:07 selector.py:115] Using XFormers backend.
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 0% Completed | 0/26 [00:00<?, ?it/s]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 4% Completed | 1/26 [00:01<00:45, 1.84s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 8% Completed | 2/26 [00:02<00:25, 1.06s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 12% Completed | 3/26 [00:04<00:32, 1.40s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 15% Completed | 4/26 [00:06<00:35, 1.62s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 19% Completed | 5/26 [00:07<00:29, 1.39s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 23% Completed | 6/26 [00:09<00:31, 1.60s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 27% Completed | 7/26 [00:09<00:23, 1.22s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 31% Completed | 8/26 [00:11<00:24, 1.39s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 35% Completed | 9/26 [00:12<00:21, 1.26s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 38% Completed | 10/26 [00:14<00:23, 1.48s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 42% Completed | 11/26 [00:14<00:16, 1.12s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 46% Completed | 12/26 [00:16<00:18, 1.35s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 50% Completed | 13/26 [00:18<00:19, 1.53s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 54% Completed | 14/26 [00:20<00:20, 1.72s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 58% Completed | 15/26 [00:21<00:15, 1.44s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 62% Completed | 16/26 [00:23<00:17, 1.72s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 65% Completed | 17/26 [00:24<00:12, 1.34s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 69% Completed | 18/26 [00:25<00:11, 1.48s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 73% Completed | 19/26 [00:27<00:10, 1.54s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 77% Completed | 20/26 [00:30<00:10, 1.81s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 81% Completed | 21/26 [00:31<00:07, 1.58s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 85% Completed | 22/26 [00:31<00:05, 1.34s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 88% Completed | 23/26 [00:34<00:04, 1.66s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 92% Completed | 24/26 [00:36<00:03, 1.75s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 96% Completed | 25/26 [00:38<00:01, 1.81s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 100% Completed | 26/26 [00:38<00:00, 1.42s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 100% Completed | 26/26 [00:38<00:00, 1.49s/it]
|
||||||
|
|
||||||
|
[1;36m(VllmWorkerProcess pid=12140)[0;0m INFO 07-14 11:58:46 model_runner.py:1123] Loading model weights took 16.2303 GB
|
||||||
|
[1;36m(VllmWorkerProcess pid=12141)[0;0m INFO 07-14 11:58:46 model_runner.py:1123] Loading model weights took 16.2303 GB
|
||||||
|
[1;36m(VllmWorkerProcess pid=12142)[0;0m INFO 07-14 11:58:46 model_runner.py:1123] Loading model weights took 16.2303 GB
|
||||||
|
INFO 07-14 11:58:46 model_runner.py:1123] Loading model weights took 16.2303 GB
|
||||||
|
INFO 07-14 11:58:54 distributed_gpu_executor.py:57] # GPU blocks: 21100, # CPU blocks: 6553
|
||||||
|
INFO 07-14 11:58:54 distributed_gpu_executor.py:61] Maximum concurrency for 100000 tokens per request: 3.38x
|
||||||
|
INFO 07-14 11:58:59 serving_chat.py:79] "auto" tool choice has been enabled please note that while the parallel_tool_calls client option is preset for compatibility reasons, it will be ignored.
|
||||||
|
INFO 07-14 11:58:59 serving_chat.py:101] Reasoning parser 'qwen3' enabled.
|
||||||
|
WARNING 07-14 11:58:59 serving_embedding.py:199] embedding_mode is False. Embedding API will not work.
|
||||||
|
INFO 07-14 11:58:59 launcher.py:19] Available routes are:
|
||||||
|
INFO 07-14 11:58:59 launcher.py:27] Route: /openapi.json, Methods: HEAD, GET
|
||||||
|
INFO 07-14 11:58:59 launcher.py:27] Route: /docs, Methods: HEAD, GET
|
||||||
|
INFO 07-14 11:58:59 launcher.py:27] Route: /docs/oauth2-redirect, Methods: HEAD, GET
|
||||||
|
INFO 07-14 11:58:59 launcher.py:27] Route: /redoc, Methods: HEAD, GET
|
||||||
|
INFO 07-14 11:58:59 launcher.py:27] Route: /health, Methods: GET
|
||||||
|
INFO 07-14 11:58:59 launcher.py:27] Route: /tokenize, Methods: POST
|
||||||
|
INFO 07-14 11:58:59 launcher.py:27] Route: /detokenize, Methods: POST
|
||||||
|
INFO 07-14 11:58:59 launcher.py:27] Route: /v1/models, Methods: GET
|
||||||
|
INFO 07-14 11:58:59 launcher.py:27] Route: /version, Methods: GET
|
||||||
|
INFO 07-14 11:58:59 launcher.py:27] Route: /v1/chat/completions, Methods: POST
|
||||||
|
INFO 07-14 11:58:59 launcher.py:27] Route: /v1/completions, Methods: POST
|
||||||
|
INFO 07-14 11:58:59 launcher.py:27] Route: /v1/embeddings, Methods: POST
|
||||||
|
INFO: Started server process [11801]
|
||||||
|
INFO: Waiting for application startup.
|
||||||
|
INFO: Application startup complete.
|
||||||
|
INFO: Uvicorn running on socket ('0.0.0.0', 1111) (Press CTRL+C to quit)
|
||||||
|
INFO 07-14 11:59:09 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:59:09 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:59:19 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:59:19 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:59:29 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:59:29 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:59:39 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:59:39 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:59:49 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:59:49 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:59:59 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:59:59 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:00:09 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:00:09 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:00:19 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:00:19 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO: 127.0.0.1:37318 - "GET /health HTTP/1.1" 200 OK
|
||||||
|
INFO 07-14 12:00:29 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:00:29 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:00:39 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:00:39 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:00:49 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:00:49 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:00:59 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:00:59 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:01:09 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:01:09 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:01:19 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:01:19 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:01:29 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:01:29 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:01:39 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:01:39 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO: 127.0.0.1:40612 - "GET /health HTTP/1.1" 200 OK
|
||||||
|
INFO 07-14 12:01:49 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:01:49 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:01:59 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:01:59 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:02:09 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:02:09 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:02:19 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:02:19 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO: 127.0.0.1:49042 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||||
|
/usr/local/lib/python3.10/site-packages/pyairports/airports.py:1: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81.
|
||||||
|
from pkg_resources import resource_string
|
||||||
|
INFO 07-14 12:02:27 metrics.py:345] Avg prompt throughput: 4.6 tokens/s, Avg generation throughput: 0.1 tokens/s, Running: 1 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:02:27 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:02:28 qwen3_5.py:97] [ENGINEX_PROFILE_QWEN] rank=0 steps=8 mode=decode prefill.layer.mlp: total=5627.88ms avg=70.348ms n=80 | prefill.moe.routed_total: total=5336.07ms avg=66.701ms n=80 | prefill.moe.routed_prefill_experts: total=5239.90ms avg=65.499ms n=80 | prefill.layer.linear_attention: total=2944.76ms avg=49.079ms n=60 | prefill.layer.full_attention: total=481.27ms avg=24.063ms n=20 | decode.layer.mlp: total=396.37ms avg=1.652ms n=240 | prefill.full_attn.paged_attention: total=371.38ms avg=18.569ms n=20 | decode.layer.linear_attention: total=253.48ms avg=1.408ms n=180 | decode.moe.routed_total: total=213.18ms avg=0.888ms n=240 | prefill.moe.tp_all_reduce: total=177.55ms avg=2.219ms n=80 | decode.moe.routed_decode_experts: total=146.93ms avg=0.612ms n=240 | prefill.layer.input_norm: total=108.49ms avg=1.356ms n=80 | prefill.layer.post_attn_norm: total=101.60ms avg=1.270ms n=80 | decode.layer.full_attention: total=95.56ms avg=1.593ms n=60 | prefill.moe.routing_topk: total=91.04ms avg=1.138ms n=80 | prefill.moe.shared_expert: total=74.97ms avg=0.937ms n=80 | decode.moe.tp_all_reduce: total=73.79ms avg=0.307ms n=240 | decode.moe.shared_expert: total=57.71ms avg=0.240ms n=240 | decode.moe.routing_topk: total=52.85ms avg=0.220ms n=240 | decode.layer.input_norm: total=50.83ms avg=0.212ms n=240 | decode.layer.post_attn_norm: total=50.66ms avg=0.211ms n=240 | prefill.full_attn.gate_o_proj: total=47.06ms avg=2.353ms n=20 | prefill.full_attn.qkv_proj: total=33.28ms avg=1.664ms n=20 | decode.full_attn.paged_attention: total=27.86ms avg=0.464ms n=60 | prefill.full_attn.norm_rope: total=27.44ms avg=1.372ms n=20 | decode.full_attn.gate_o_proj: total=25.08ms avg=0.418ms n=60 | decode.full_attn.norm_rope: total=24.31ms avg=0.405ms n=60 | prefill.moe.gate: total=19.43ms avg=0.243ms n=80 | decode.moe.gate: total=16.86ms avg=0.070ms n=240 | decode.full_attn.qkv_proj: total=12.79ms avg=0.213ms n=60 | prefill.moe.combine: total=11.04ms avg=0.138ms n=80 | decode.moe.combine: total=10.02ms avg=0.042ms n=240
|
||||||
|
[1;36m(VllmWorkerProcess pid=12141)[0;0m INFO 07-14 12:02:28 qwen3_5.py:97] [ENGINEX_PROFILE_QWEN] rank=2 steps=8 mode=decode prefill.layer.mlp: total=5629.85ms avg=70.373ms n=80 | prefill.moe.routed_total: total=5306.20ms avg=66.327ms n=80 | prefill.moe.routed_prefill_experts: total=5215.66ms avg=65.196ms n=80 | prefill.layer.linear_attention: total=2945.59ms avg=49.093ms n=60 | prefill.layer.full_attention: total=481.58ms avg=24.079ms n=20 | decode.layer.mlp: total=397.69ms avg=1.657ms n=240 | prefill.full_attn.paged_attention: total=370.72ms avg=18.536ms n=20 | decode.layer.linear_attention: total=253.55ms avg=1.409ms n=180 | decode.moe.routed_total: total=212.43ms avg=0.885ms n=240 | prefill.moe.tp_all_reduce: total=209.76ms avg=2.622ms n=80 | decode.moe.routed_decode_experts: total=146.82ms avg=0.612ms n=240 | prefill.layer.input_norm: total=106.27ms avg=1.328ms n=80 | prefill.layer.post_attn_norm: total=100.86ms avg=1.261ms n=80 | decode.layer.full_attention: total=95.54ms avg=1.592ms n=60 | prefill.moe.routing_topk: total=85.44ms avg=1.068ms n=80 | decode.moe.tp_all_reduce: total=76.40ms avg=0.318ms n=240 | prefill.moe.shared_expert: total=74.77ms avg=0.935ms n=80 | decode.moe.shared_expert: total=57.43ms avg=0.239ms n=240 | decode.moe.routing_topk: total=52.43ms avg=0.218ms n=240 | decode.layer.post_attn_norm: total=50.27ms avg=0.209ms n=240 | decode.layer.input_norm: total=49.99ms avg=0.208ms n=240 | prefill.full_attn.gate_o_proj: total=48.41ms avg=2.420ms n=20 | prefill.full_attn.qkv_proj: total=33.20ms avg=1.660ms n=20 | decode.full_attn.paged_attention: total=27.29ms avg=0.455ms n=60 | prefill.full_attn.norm_rope: total=27.29ms avg=1.364ms n=20 | decode.full_attn.gate_o_proj: total=25.68ms avg=0.428ms n=60 | decode.full_attn.norm_rope: total=24.27ms avg=0.404ms n=60 | prefill.moe.gate: total=19.33ms avg=0.242ms n=80 | decode.moe.gate: total=16.68ms avg=0.069ms n=240 | decode.full_attn.qkv_proj: total=12.81ms avg=0.214ms n=60 | prefill.moe.combine: total=10.99ms avg=0.137ms n=80 | decode.moe.combine: total=10.03ms avg=0.042ms n=240
|
||||||
|
[1;36m(VllmWorkerProcess pid=12142)[0;0m INFO 07-14 12:02:28 qwen3_5.py:97] [ENGINEX_PROFILE_QWEN] rank=3 steps=8 mode=decode prefill.layer.mlp: total=5628.82ms avg=70.360ms n=80 | prefill.moe.routed_total: total=5306.10ms avg=66.326ms n=80 | prefill.moe.routed_prefill_experts: total=5214.38ms avg=65.180ms n=80 | prefill.layer.linear_attention: total=2944.17ms avg=49.070ms n=60 | prefill.layer.full_attention: total=481.30ms avg=24.065ms n=20 | decode.layer.mlp: total=398.03ms avg=1.658ms n=240 | prefill.full_attn.paged_attention: total=370.76ms avg=18.538ms n=20 | decode.layer.linear_attention: total=254.00ms avg=1.411ms n=180 | decode.moe.routed_total: total=211.60ms avg=0.882ms n=240 | prefill.moe.tp_all_reduce: total=209.17ms avg=2.615ms n=80 | decode.moe.routed_decode_experts: total=146.21ms avg=0.609ms n=240 | prefill.layer.input_norm: total=108.67ms avg=1.358ms n=80 | prefill.layer.post_attn_norm: total=101.07ms avg=1.263ms n=80 | decode.layer.full_attention: total=95.75ms avg=1.596ms n=60 | prefill.moe.routing_topk: total=86.53ms avg=1.082ms n=80 | decode.moe.tp_all_reduce: total=78.58ms avg=0.327ms n=240 | prefill.moe.shared_expert: total=74.52ms avg=0.931ms n=80 | decode.moe.shared_expert: total=56.73ms avg=0.236ms n=240 | decode.moe.routing_topk: total=52.04ms avg=0.217ms n=240 | decode.layer.post_attn_norm: total=49.95ms avg=0.208ms n=240 | decode.layer.input_norm: total=49.23ms avg=0.205ms n=240 | prefill.full_attn.gate_o_proj: total=48.03ms avg=2.402ms n=20 | prefill.full_attn.qkv_proj: total=33.18ms avg=1.659ms n=20 | decode.full_attn.paged_attention: total=27.44ms avg=0.457ms n=60 | prefill.full_attn.norm_rope: total=27.33ms avg=1.366ms n=20 | decode.full_attn.gate_o_proj: total=26.08ms avg=0.435ms n=60 | decode.full_attn.norm_rope: total=24.07ms avg=0.401ms n=60 | prefill.moe.gate: total=19.19ms avg=0.240ms n=80 | decode.moe.gate: total=16.60ms avg=0.069ms n=240 | decode.full_attn.qkv_proj: total=12.70ms avg=0.212ms n=60 | prefill.moe.combine: total=10.96ms avg=0.137ms n=80 | decode.moe.combine: total=9.96ms avg=0.042ms n=240
|
||||||
|
[1;36m(VllmWorkerProcess pid=12140)[0;0m INFO 07-14 12:02:28 qwen3_5.py:97] [ENGINEX_PROFILE_QWEN] rank=1 steps=8 mode=decode prefill.layer.mlp: total=5629.78ms avg=70.372ms n=80 | prefill.moe.routed_total: total=5372.03ms avg=67.150ms n=80 | prefill.moe.routed_prefill_experts: total=5280.95ms avg=66.012ms n=80 | prefill.layer.linear_attention: total=2944.91ms avg=49.082ms n=60 | prefill.layer.full_attention: total=481.27ms avg=24.064ms n=20 | decode.layer.mlp: total=395.86ms avg=1.649ms n=240 | prefill.full_attn.paged_attention: total=371.47ms avg=18.574ms n=20 | decode.layer.linear_attention: total=253.23ms avg=1.407ms n=180 | decode.moe.routed_total: total=213.96ms avg=0.892ms n=240 | decode.moe.routed_decode_experts: total=147.30ms avg=0.614ms n=240 | prefill.moe.tp_all_reduce: total=143.85ms avg=1.798ms n=80 | prefill.layer.input_norm: total=107.20ms avg=1.340ms n=80 | prefill.layer.post_attn_norm: total=101.07ms avg=1.263ms n=80 | decode.layer.full_attention: total=95.48ms avg=1.591ms n=60 | prefill.moe.routing_topk: total=86.03ms avg=1.075ms n=80 | prefill.moe.shared_expert: total=74.58ms avg=0.932ms n=80 | decode.moe.tp_all_reduce: total=70.90ms avg=0.295ms n=240 | decode.moe.shared_expert: total=58.24ms avg=0.243ms n=240 | decode.moe.routing_topk: total=53.61ms avg=0.223ms n=240 | decode.layer.input_norm: total=51.24ms avg=0.214ms n=240 | decode.layer.post_attn_norm: total=51.23ms avg=0.213ms n=240 | prefill.full_attn.gate_o_proj: total=47.30ms avg=2.365ms n=20 | prefill.full_attn.qkv_proj: total=33.16ms avg=1.658ms n=20 | decode.full_attn.paged_attention: total=27.88ms avg=0.465ms n=60 | prefill.full_attn.norm_rope: total=27.36ms avg=1.368ms n=20 | decode.full_attn.gate_o_proj: total=24.87ms avg=0.414ms n=60 | decode.full_attn.norm_rope: total=24.44ms avg=0.407ms n=60 | prefill.moe.gate: total=19.37ms avg=0.242ms n=80 | decode.moe.gate: total=17.31ms avg=0.072ms n=240 | decode.full_attn.qkv_proj: total=12.87ms avg=0.215ms n=60 | prefill.moe.combine: total=11.14ms avg=0.139ms n=80 | decode.moe.combine: total=10.87ms avg=0.045ms n=240
|
||||||
|
INFO 07-14 12:02:28 model_runner.py:114] [ENGINEX_PROFILE_MODEL_RUNNER] steps=8 mode=decode prompt.model_forward: total=9427.11ms avg=4713.555ms n=2 | decode.model_forward: total=876.39ms avg=146.065ms n=6 | prompt.compute_logits: total=204.51ms avg=102.255ms n=2 | prompt.sample: total=34.98ms avg=17.492ms n=2 | decode.compute_logits: total=7.07ms avg=1.178ms n=6 | decode.sample: total=6.06ms avg=1.010ms n=6 | decode.attn_begin_forward: total=0.10ms avg=0.016ms n=6 | prompt.attn_begin_forward: total=0.03ms avg=0.017ms n=2
|
||||||
|
[1;36m(VllmWorkerProcess pid=12141)[0;0m INFO 07-14 12:02:29 qwen3_5.py:97] [ENGINEX_PROFILE_QWEN] rank=2 steps=16 mode=decode prefill.layer.mlp: total=5629.85ms avg=70.373ms n=80 | prefill.moe.routed_total: total=5306.20ms avg=66.327ms n=80 | prefill.moe.routed_prefill_experts: total=5215.66ms avg=65.196ms n=80 | prefill.layer.linear_attention: total=2945.59ms avg=49.093ms n=60 | decode.layer.mlp: total=925.00ms avg=1.652ms n=560 | decode.layer.linear_attention: total=591.69ms avg=1.409ms n=420 | decode.moe.routed_total: total=493.80ms avg=0.882ms n=560 | prefill.layer.full_attention: total=481.58ms avg=24.079ms n=20 | prefill.full_attn.paged_attention: total=370.72ms avg=18.536ms n=20 | decode.moe.routed_decode_experts: total=341.90ms avg=0.611ms n=560 | prefill.moe.tp_all_reduce: total=209.76ms avg=2.622ms n=80 | decode.layer.full_attention: total=207.84ms avg=1.485ms n=140 | decode.moe.tp_all_reduce: total=178.18ms avg=0.318ms n=560 | decode.moe.shared_expert: total=133.05ms avg=0.238ms n=560 | decode.moe.routing_topk: total=121.08ms avg=0.216ms n=560 | decode.layer.input_norm: total=116.38ms avg=0.208ms n=560 | decode.layer.post_attn_norm: total=114.92ms avg=0.205ms n=560 | prefill.layer.input_norm: total=106.27ms avg=1.328ms n=80 | prefill.layer.post_attn_norm: total=100.86ms avg=1.261ms n=80 | prefill.moe.routing_topk: total=85.44ms avg=1.068ms n=80 | prefill.moe.shared_expert: total=74.77ms avg=0.935ms n=80 | decode.full_attn.gate_o_proj: total=60.31ms avg=0.431ms n=140 | decode.full_attn.norm_rope: total=56.45ms avg=0.403ms n=140 | decode.full_attn.paged_attention: total=48.46ms avg=0.346ms n=140 | prefill.full_attn.gate_o_proj: total=48.41ms avg=2.420ms n=20 | decode.moe.gate: total=39.64ms avg=0.071ms n=560 | prefill.full_attn.qkv_proj: total=33.20ms avg=1.660ms n=20 | decode.full_attn.qkv_proj: total=29.77ms avg=0.213ms n=140 | prefill.full_attn.norm_rope: total=27.29ms avg=1.364ms n=20 | decode.moe.combine: total=22.66ms avg=0.040ms n=560 | prefill.moe.gate: total=19.33ms avg=0.242ms n=80 | prefill.moe.combine: total=10.99ms avg=0.137ms n=80
|
||||||
|
[1;36m(VllmWorkerProcess pid=12142)[0;0m INFO 07-14 12:02:29 qwen3_5.py:97] [ENGINEX_PROFILE_QWEN] rank=3 steps=16 mode=decode prefill.layer.mlp: total=5628.82ms avg=70.360ms n=80 | prefill.moe.routed_total: total=5306.10ms avg=66.326ms n=80 | prefill.moe.routed_prefill_experts: total=5214.38ms avg=65.180ms n=80 | prefill.layer.linear_attention: total=2944.17ms avg=49.070ms n=60 | decode.layer.mlp: total=925.51ms avg=1.653ms n=560 | decode.layer.linear_attention: total=592.23ms avg=1.410ms n=420 | decode.moe.routed_total: total=491.76ms avg=0.878ms n=560 | prefill.layer.full_attention: total=481.30ms avg=24.065ms n=20 | prefill.full_attn.paged_attention: total=370.76ms avg=18.538ms n=20 | decode.moe.routed_decode_experts: total=339.95ms avg=0.607ms n=560 | prefill.moe.tp_all_reduce: total=209.17ms avg=2.615ms n=80 | decode.layer.full_attention: total=208.09ms avg=1.486ms n=140 | decode.moe.tp_all_reduce: total=182.78ms avg=0.326ms n=560 | decode.moe.shared_expert: total=131.57ms avg=0.235ms n=560 | decode.moe.routing_topk: total=120.78ms avg=0.216ms n=560 | decode.layer.input_norm: total=114.97ms avg=0.205ms n=560 | decode.layer.post_attn_norm: total=114.38ms avg=0.204ms n=560 | prefill.layer.input_norm: total=108.67ms avg=1.358ms n=80 | prefill.layer.post_attn_norm: total=101.07ms avg=1.263ms n=80 | prefill.moe.routing_topk: total=86.53ms avg=1.082ms n=80 | prefill.moe.shared_expert: total=74.52ms avg=0.931ms n=80 | decode.full_attn.gate_o_proj: total=60.87ms avg=0.435ms n=140 | decode.full_attn.norm_rope: total=56.11ms avg=0.401ms n=140 | decode.full_attn.paged_attention: total=48.71ms avg=0.348ms n=140 | prefill.full_attn.gate_o_proj: total=48.03ms avg=2.402ms n=20 | decode.moe.gate: total=39.45ms avg=0.070ms n=560 | prefill.full_attn.qkv_proj: total=33.18ms avg=1.659ms n=20 | decode.full_attn.qkv_proj: total=29.58ms avg=0.211ms n=140 | prefill.full_attn.norm_rope: total=27.33ms avg=1.366ms n=20 | decode.moe.combine: total=22.44ms avg=0.040ms n=560 | prefill.moe.gate: total=19.19ms avg=0.240ms n=80 | prefill.moe.combine: total=10.96ms avg=0.137ms n=80
|
||||||
|
INFO 07-14 12:02:29 qwen3_5.py:97] [ENGINEX_PROFILE_QWEN] rank=0 steps=16 mode=decode prefill.layer.mlp: total=5627.88ms avg=70.348ms n=80 | prefill.moe.routed_total: total=5336.07ms avg=66.701ms n=80 | prefill.moe.routed_prefill_experts: total=5239.90ms avg=65.499ms n=80 | prefill.layer.linear_attention: total=2944.76ms avg=49.079ms n=60 | decode.layer.mlp: total=920.78ms avg=1.644ms n=560 | decode.layer.linear_attention: total=591.72ms avg=1.409ms n=420 | decode.moe.routed_total: total=496.91ms avg=0.887ms n=560 | prefill.layer.full_attention: total=481.27ms avg=24.063ms n=20 | prefill.full_attn.paged_attention: total=371.38ms avg=18.569ms n=20 | decode.moe.routed_decode_experts: total=342.49ms avg=0.612ms n=560 | decode.layer.full_attention: total=207.43ms avg=1.482ms n=140 | prefill.moe.tp_all_reduce: total=177.55ms avg=2.219ms n=80 | decode.moe.tp_all_reduce: total=168.09ms avg=0.300ms n=560 | decode.moe.shared_expert: total=134.68ms avg=0.241ms n=560 | decode.moe.routing_topk: total=122.92ms avg=0.219ms n=560 | decode.layer.input_norm: total=118.83ms avg=0.212ms n=560 | decode.layer.post_attn_norm: total=116.47ms avg=0.208ms n=560 | prefill.layer.input_norm: total=108.49ms avg=1.356ms n=80 | prefill.layer.post_attn_norm: total=101.60ms avg=1.270ms n=80 | prefill.moe.routing_topk: total=91.04ms avg=1.138ms n=80 | prefill.moe.shared_expert: total=74.97ms avg=0.937ms n=80 | decode.full_attn.gate_o_proj: total=58.00ms avg=0.414ms n=140 | decode.full_attn.norm_rope: total=56.74ms avg=0.405ms n=140 | decode.full_attn.paged_attention: total=49.77ms avg=0.355ms n=140 | prefill.full_attn.gate_o_proj: total=47.06ms avg=2.353ms n=20 | decode.moe.gate: total=40.44ms avg=0.072ms n=560 | prefill.full_attn.qkv_proj: total=33.28ms avg=1.664ms n=20 | decode.full_attn.qkv_proj: total=30.03ms avg=0.214ms n=140 | prefill.full_attn.norm_rope: total=27.44ms avg=1.372ms n=20 | decode.moe.combine: total=22.89ms avg=0.041ms n=560 | prefill.moe.gate: total=19.43ms avg=0.243ms n=80 | prefill.moe.combine: total=11.04ms avg=0.138ms n=80
|
||||||
|
[1;36m(VllmWorkerProcess pid=12140)[0;0m INFO 07-14 12:02:29 qwen3_5.py:97] [ENGINEX_PROFILE_QWEN] rank=1 steps=16 mode=decode prefill.layer.mlp: total=5629.78ms avg=70.372ms n=80 | prefill.moe.routed_total: total=5372.03ms avg=67.150ms n=80 | prefill.moe.routed_prefill_experts: total=5280.95ms avg=66.012ms n=80 | prefill.layer.linear_attention: total=2944.91ms avg=49.082ms n=60 | decode.layer.mlp: total=920.89ms avg=1.644ms n=560 | decode.layer.linear_attention: total=591.73ms avg=1.409ms n=420 | decode.moe.routed_total: total=497.21ms avg=0.888ms n=560 | prefill.layer.full_attention: total=481.27ms avg=24.064ms n=20 | prefill.full_attn.paged_attention: total=371.47ms avg=18.574ms n=20 | decode.moe.routed_decode_experts: total=343.09ms avg=0.613ms n=560 | decode.layer.full_attention: total=207.57ms avg=1.483ms n=140 | decode.moe.tp_all_reduce: total=166.13ms avg=0.297ms n=560 | prefill.moe.tp_all_reduce: total=143.85ms avg=1.798ms n=80 | decode.moe.shared_expert: total=134.40ms avg=0.240ms n=560 | decode.moe.routing_topk: total=123.62ms avg=0.221ms n=560 | decode.layer.input_norm: total=118.60ms avg=0.212ms n=560 | decode.layer.post_attn_norm: total=116.99ms avg=0.209ms n=560 | prefill.layer.input_norm: total=107.20ms avg=1.340ms n=80 | prefill.layer.post_attn_norm: total=101.07ms avg=1.263ms n=80 | prefill.moe.routing_topk: total=86.03ms avg=1.075ms n=80 | prefill.moe.shared_expert: total=74.58ms avg=0.932ms n=80 | decode.full_attn.gate_o_proj: total=58.25ms avg=0.416ms n=140 | decode.full_attn.norm_rope: total=56.87ms avg=0.406ms n=140 | decode.full_attn.paged_attention: total=49.87ms avg=0.356ms n=140 | prefill.full_attn.gate_o_proj: total=47.30ms avg=2.365ms n=20 | decode.moe.gate: total=41.26ms avg=0.074ms n=560 | prefill.full_attn.qkv_proj: total=33.16ms avg=1.658ms n=20 | decode.full_attn.qkv_proj: total=29.95ms avg=0.214ms n=140 | prefill.full_attn.norm_rope: total=27.36ms avg=1.368ms n=20 | decode.moe.combine: total=24.68ms avg=0.044ms n=560 | prefill.moe.gate: total=19.37ms avg=0.242ms n=80 | prefill.moe.combine: total=11.14ms avg=0.139ms n=80
|
||||||
|
INFO 07-14 12:02:29 model_runner.py:114] [ENGINEX_PROFILE_MODEL_RUNNER] steps=16 mode=decode prompt.model_forward: total=9427.11ms avg=4713.555ms n=2 | decode.model_forward: total=2023.26ms avg=144.518ms n=14 | prompt.compute_logits: total=204.51ms avg=102.255ms n=2 | prompt.sample: total=34.98ms avg=17.492ms n=2 | decode.compute_logits: total=16.28ms avg=1.163ms n=14 | decode.sample: total=14.08ms avg=1.006ms n=14 | decode.attn_begin_forward: total=0.23ms avg=0.016ms n=14 | prompt.attn_begin_forward: total=0.03ms avg=0.017ms n=2
|
||||||
|
[1;36m(VllmWorkerProcess pid=12142)[0;0m INFO 07-14 12:02:30 qwen3_5.py:97] [ENGINEX_PROFILE_QWEN] rank=3 steps=24 mode=decode prefill.layer.mlp: total=5628.82ms avg=70.360ms n=80 | prefill.moe.routed_total: total=5306.10ms avg=66.326ms n=80 | prefill.moe.routed_prefill_experts: total=5214.38ms avg=65.180ms n=80 | prefill.layer.linear_attention: total=2944.17ms avg=49.070ms n=60 | decode.layer.mlp: total=1464.30ms avg=1.664ms n=880 | decode.layer.linear_attention: total=941.70ms avg=1.427ms n=660 | decode.moe.routed_total: total=775.83ms avg=0.882ms n=880 | decode.moe.routed_decode_experts: total=535.16ms avg=0.608ms n=880 | prefill.layer.full_attention: total=481.30ms avg=24.065ms n=20 | prefill.full_attn.paged_attention: total=370.76ms avg=18.538ms n=20 | decode.layer.full_attention: total=323.02ms avg=1.468ms n=220 | decode.moe.tp_all_reduce: total=291.99ms avg=0.332ms n=880 | prefill.moe.tp_all_reduce: total=209.17ms avg=2.615ms n=80 | decode.moe.shared_expert: total=207.50ms avg=0.236ms n=880 | decode.moe.routing_topk: total=191.33ms avg=0.217ms n=880 | decode.layer.input_norm: total=184.46ms avg=0.210ms n=880 | decode.layer.post_attn_norm: total=180.84ms avg=0.205ms n=880 | prefill.layer.input_norm: total=108.67ms avg=1.358ms n=80 | prefill.layer.post_attn_norm: total=101.07ms avg=1.263ms n=80 | decode.full_attn.gate_o_proj: total=96.14ms avg=0.437ms n=220 | decode.full_attn.norm_rope: total=88.66ms avg=0.403ms n=220 | prefill.moe.routing_topk: total=86.53ms avg=1.082ms n=80 | prefill.moe.shared_expert: total=74.52ms avg=0.931ms n=80 | decode.full_attn.paged_attention: total=71.05ms avg=0.323ms n=220 | decode.moe.gate: total=61.77ms avg=0.070ms n=880 | prefill.full_attn.gate_o_proj: total=48.03ms avg=2.402ms n=20 | decode.full_attn.qkv_proj: total=46.77ms avg=0.213ms n=220 | decode.moe.combine: total=35.39ms avg=0.040ms n=880 | prefill.full_attn.qkv_proj: total=33.18ms avg=1.659ms n=20 | prefill.full_attn.norm_rope: total=27.33ms avg=1.366ms n=20 | prefill.moe.gate: total=19.19ms avg=0.240ms n=80 | prefill.moe.combine: total=10.96ms avg=0.137ms n=80
|
||||||
|
[1;36m(VllmWorkerProcess pid=12141)[0;0m INFO 07-14 12:02:30 qwen3_5.py:97] [ENGINEX_PROFILE_QWEN] rank=2 steps=24 mode=decode prefill.layer.mlp: total=5629.85ms avg=70.373ms n=80 | prefill.moe.routed_total: total=5306.20ms avg=66.327ms n=80 | prefill.moe.routed_prefill_experts: total=5215.66ms avg=65.196ms n=80 | prefill.layer.linear_attention: total=2945.59ms avg=49.093ms n=60 | decode.layer.mlp: total=1463.70ms avg=1.663ms n=880 | decode.layer.linear_attention: total=943.19ms avg=1.429ms n=660 | decode.moe.routed_total: total=779.57ms avg=0.886ms n=880 | decode.moe.routed_decode_experts: total=538.76ms avg=0.612ms n=880 | prefill.layer.full_attention: total=481.58ms avg=24.079ms n=20 | prefill.full_attn.paged_attention: total=370.72ms avg=18.536ms n=20 | decode.layer.full_attention: total=322.62ms avg=1.466ms n=220 | decode.moe.tp_all_reduce: total=283.04ms avg=0.322ms n=880 | decode.moe.shared_expert: total=210.88ms avg=0.240ms n=880 | prefill.moe.tp_all_reduce: total=209.76ms avg=2.622ms n=80 | decode.moe.routing_topk: total=191.55ms avg=0.218ms n=880 | decode.layer.input_norm: total=184.49ms avg=0.210ms n=880 | decode.layer.post_attn_norm: total=181.60ms avg=0.206ms n=880 | prefill.layer.input_norm: total=106.27ms avg=1.328ms n=80 | prefill.layer.post_attn_norm: total=100.86ms avg=1.261ms n=80 | decode.full_attn.gate_o_proj: total=95.24ms avg=0.433ms n=220 | decode.full_attn.norm_rope: total=89.21ms avg=0.406ms n=220 | prefill.moe.routing_topk: total=85.44ms avg=1.068ms n=80 | prefill.moe.shared_expert: total=74.77ms avg=0.935ms n=80 | decode.full_attn.paged_attention: total=70.81ms avg=0.322ms n=220 | decode.moe.gate: total=62.05ms avg=0.071ms n=880 | prefill.full_attn.gate_o_proj: total=48.41ms avg=2.420ms n=20 | decode.full_attn.qkv_proj: total=46.98ms avg=0.214ms n=220 | decode.moe.combine: total=35.85ms avg=0.041ms n=880 | prefill.full_attn.qkv_proj: total=33.20ms avg=1.660ms n=20 | prefill.full_attn.norm_rope: total=27.29ms avg=1.364ms n=20 | prefill.moe.gate: total=19.33ms avg=0.242ms n=80 | prefill.moe.combine: total=10.99ms avg=0.137ms n=80
|
||||||
|
INFO 07-14 12:02:30 qwen3_5.py:97] [ENGINEX_PROFILE_QWEN] rank=0 steps=24 mode=decode prefill.layer.mlp: total=5627.88ms avg=70.348ms n=80 | prefill.moe.routed_total: total=5336.07ms avg=66.701ms n=80 | prefill.moe.routed_prefill_experts: total=5239.90ms avg=65.499ms n=80 | prefill.layer.linear_attention: total=2944.76ms avg=49.079ms n=60 | decode.layer.mlp: total=1456.57ms avg=1.655ms n=880 | decode.layer.linear_attention: total=942.63ms avg=1.428ms n=660 | decode.moe.routed_total: total=785.65ms avg=0.893ms n=880 | decode.moe.routed_decode_experts: total=540.24ms avg=0.614ms n=880 | prefill.layer.full_attention: total=481.27ms avg=24.063ms n=20 | prefill.full_attn.paged_attention: total=371.38ms avg=18.569ms n=20 | decode.layer.full_attention: total=322.06ms avg=1.464ms n=220 | decode.moe.tp_all_reduce: total=265.92ms avg=0.302ms n=880 | decode.moe.shared_expert: total=213.28ms avg=0.242ms n=880 | decode.moe.routing_topk: total=195.06ms avg=0.222ms n=880 | decode.layer.input_norm: total=188.81ms avg=0.215ms n=880 | decode.layer.post_attn_norm: total=184.60ms avg=0.210ms n=880 | prefill.moe.tp_all_reduce: total=177.55ms avg=2.219ms n=80 | prefill.layer.input_norm: total=108.49ms avg=1.356ms n=80 | prefill.layer.post_attn_norm: total=101.60ms avg=1.270ms n=80 | decode.full_attn.gate_o_proj: total=91.78ms avg=0.417ms n=220 | prefill.moe.routing_topk: total=91.04ms avg=1.138ms n=80 | decode.full_attn.norm_rope: total=89.66ms avg=0.408ms n=220 | prefill.moe.shared_expert: total=74.97ms avg=0.937ms n=80 | decode.full_attn.paged_attention: total=72.73ms avg=0.331ms n=220 | decode.moe.gate: total=63.37ms avg=0.072ms n=880 | decode.full_attn.qkv_proj: total=47.46ms avg=0.216ms n=220 | prefill.full_attn.gate_o_proj: total=47.06ms avg=2.353ms n=20 | decode.moe.combine: total=36.03ms avg=0.041ms n=880 | prefill.full_attn.qkv_proj: total=33.28ms avg=1.664ms n=20 | prefill.full_attn.norm_rope: total=27.44ms avg=1.372ms n=20 | prefill.moe.gate: total=19.43ms avg=0.243ms n=80 | prefill.moe.combine: total=11.04ms avg=0.138ms n=80
|
||||||
|
[1;36m(VllmWorkerProcess pid=12140)[0;0m INFO 07-14 12:02:30 qwen3_5.py:97] [ENGINEX_PROFILE_QWEN] rank=1 steps=24 mode=decode prefill.layer.mlp: total=5629.78ms avg=70.372ms n=80 | prefill.moe.routed_total: total=5372.03ms avg=67.150ms n=80 | prefill.moe.routed_prefill_experts: total=5280.95ms avg=66.012ms n=80 | prefill.layer.linear_attention: total=2944.91ms avg=49.082ms n=60 | decode.layer.mlp: total=1457.27ms avg=1.656ms n=880 | decode.layer.linear_attention: total=943.04ms avg=1.429ms n=660 | decode.moe.routed_total: total=783.51ms avg=0.890ms n=880 | decode.moe.routed_decode_experts: total=540.23ms avg=0.614ms n=880 | prefill.layer.full_attention: total=481.27ms avg=24.064ms n=20 | prefill.full_attn.paged_attention: total=371.47ms avg=18.574ms n=20 | decode.layer.full_attention: total=322.30ms avg=1.465ms n=220 | decode.moe.tp_all_reduce: total=267.01ms avg=0.303ms n=880 | decode.moe.shared_expert: total=212.02ms avg=0.241ms n=880 | decode.moe.routing_topk: total=194.56ms avg=0.221ms n=880 | decode.layer.input_norm: total=187.87ms avg=0.213ms n=880 | decode.layer.post_attn_norm: total=185.10ms avg=0.210ms n=880 | prefill.moe.tp_all_reduce: total=143.85ms avg=1.798ms n=80 | prefill.layer.input_norm: total=107.20ms avg=1.340ms n=80 | prefill.layer.post_attn_norm: total=101.07ms avg=1.263ms n=80 | decode.full_attn.gate_o_proj: total=92.80ms avg=0.422ms n=220 | decode.full_attn.norm_rope: total=89.57ms avg=0.407ms n=220 | prefill.moe.routing_topk: total=86.03ms avg=1.075ms n=80 | prefill.moe.shared_expert: total=74.58ms avg=0.932ms n=80 | decode.full_attn.paged_attention: total=72.43ms avg=0.329ms n=220 | decode.moe.gate: total=64.64ms avg=0.073ms n=880 | decode.full_attn.qkv_proj: total=47.37ms avg=0.215ms n=220 | prefill.full_attn.gate_o_proj: total=47.30ms avg=2.365ms n=20 | decode.moe.combine: total=38.74ms avg=0.044ms n=880 | prefill.full_attn.qkv_proj: total=33.16ms avg=1.658ms n=20 | prefill.full_attn.norm_rope: total=27.36ms avg=1.368ms n=20 | prefill.moe.gate: total=19.37ms avg=0.242ms n=80 | prefill.moe.combine: total=11.14ms avg=0.139ms n=80
|
||||||
|
INFO 07-14 12:02:30 model_runner.py:114] [ENGINEX_PROFILE_MODEL_RUNNER] steps=24 mode=decode prompt.model_forward: total=9427.11ms avg=4713.555ms n=2 | decode.model_forward: total=3204.14ms avg=145.643ms n=22 | prompt.compute_logits: total=204.51ms avg=102.255ms n=2 | prompt.sample: total=34.98ms avg=17.492ms n=2 | decode.compute_logits: total=25.62ms avg=1.165ms n=22 | decode.sample: total=22.16ms avg=1.007ms n=22 | decode.attn_begin_forward: total=0.36ms avg=0.017ms n=22 | prompt.attn_begin_forward: total=0.03ms avg=0.017ms n=2
|
||||||
|
INFO 07-14 12:02:39 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 2.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:02:39 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:02:49 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:02:49 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:02:59 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:02:59 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:03:09 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:03:09 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:03:19 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:03:19 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:03:29 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:03:29 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:03:39 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:03:39 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:03:49 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:03:49 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:03:59 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:03:59 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:04:09 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:04:09 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:04:19 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:04:19 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:04:29 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:04:29 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:04:39 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:04:39 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:04:49 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:04:49 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:04:59 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:04:59 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:05:09 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:05:09 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:05:19 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:05:19 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:05:29 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:05:29 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:05:39 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:05:39 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:05:49 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:05:49 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:05:59 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:05:59 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:06:09 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:06:09 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:06:19 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:06:19 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:06:29 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:06:29 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:06:39 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:06:39 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:06:49 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:06:49 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:06:59 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:06:59 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:07:09 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:07:09 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:07:19 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:07:19 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:07:29 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:07:29 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:07:39 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:07:39 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:07:49 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:07:49 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:07:59 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:07:59 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:08:09 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:08:09 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:08:19 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:08:19 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:08:29 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:08:29 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:08:39 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:08:39 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:08:49 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:08:49 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:08:59 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:08:59 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:09:09 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:09:09 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:09:19 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:09:19 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:09:29 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:09:29 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:09:39 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:09:39 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:09:49 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:09:49 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:09:59 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:09:59 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:10:09 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:10:09 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:10:19 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:10:19 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:10:29 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:10:29 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:10:39 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:10:39 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:10:49 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:10:49 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:10:59 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:10:59 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:11:09 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:11:09 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:11:19 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:11:19 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:11:29 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:11:29 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:11:39 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:11:39 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:11:49 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:11:49 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:11:59 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:11:59 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:12:09 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:12:09 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:12:19 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:12:19 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:12:29 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:12:29 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:12:39 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:12:39 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:12:49 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:12:49 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:12:59 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:12:59 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:13:09 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:13:09 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
@@ -0,0 +1,303 @@
|
|||||||
|
/usr/local/corex/lib/python3/dist-packages/torch/cuda/__init__.py:51: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you.
|
||||||
|
import pynvml # type: ignore[import]
|
||||||
|
INFO 07-14 12:25:08 importing.py:10] Triton not installed; certain GPU-related functions will not be available.
|
||||||
|
2026-07-14 12:25:10.276354: I tensorflow/core/util/port.cc:110] oneDNN custom operations are on. You may see slightly different numerical results due to floating-point round-off errors from different computation orders. To turn them off, set the environment variable `TF_ENABLE_ONEDNN_OPTS=0`.
|
||||||
|
2026-07-14 12:25:10.327684: I tensorflow/core/platform/cpu_feature_guard.cc:182] This TensorFlow binary is optimized to use available CPU instructions in performance-critical operations.
|
||||||
|
To enable the following instructions: SSE3 SSE4.1 SSE4.2 AVX AVX2 AVX512F AVX512_VNNI AVX512_BF16 AVX_VNNI AMX_TILE AMX_INT8 AMX_BF16 FMA, in other operations, rebuild TensorFlow with the appropriate compiler flags.
|
||||||
|
WARNING:tensorflow:Deprecation warnings have been disabled. Set TF_ENABLE_DEPRECATION_WARNINGS=1 to re-enable them.
|
||||||
|
INFO 07-14 12:25:15 api_server.py:530] vLLM API server version 0.6.3
|
||||||
|
INFO 07-14 12:25:15 api_server.py:531] args: Namespace(host='0.0.0.0', port=1111, uvicorn_log_level='info', allow_credentials=False, allowed_origins=['*'], allowed_methods=['*'], allowed_headers=['*'], api_key=None, lora_modules=None, prompt_adapters=None, chat_template=None, response_role='assistant', ssl_keyfile=None, ssl_certfile=None, ssl_ca_certs=None, ssl_cert_reqs=0, root_path=None, middleware=[], return_tokens_as_token_ids=False, disable_frontend_multiprocessing=True, enable_auto_tool_choice=True, tool_call_parser='qwen3_coder', tool_parser_plugin='', reasoning_parser='qwen3', model='/root/public-storage/models/Qwen/Qwen3.6-35B-A3B', tokenizer=None, skip_tokenizer_init=False, revision=None, code_revision=None, tokenizer_revision=None, tokenizer_mode='auto', trust_remote_code=True, download_dir=None, load_format='auto', config_format='auto', dtype='auto', kv_cache_dtype='auto', quantization_param_path=None, max_model_len=100000, guided_decoding_backend='outlines', distributed_executor_backend=None, worker_use_ray=False, pipeline_parallel_size=1, tensor_parallel_size=4, max_parallel_loading_workers=None, ray_workers_use_nsight=False, block_size=16, enable_prefix_caching=True, disable_sliding_window=False, use_v2_block_manager=True, num_lookahead_slots=0, seed=0, swap_space=4, cpu_offload_gb=0, gpu_memory_utilization=0.95, num_gpu_blocks_override=None, max_num_batched_tokens=8192, max_num_seqs=2, max_logprobs=20, disable_log_stats=False, quantization=None, rope_scaling=None, rope_theta=None, enforce_eager=True, max_context_len_to_capture=None, max_seq_len_to_capture=32768, disable_custom_all_reduce=False, tokenizer_pool_size=0, tokenizer_pool_type='ray', tokenizer_pool_extra_config=None, limit_mm_per_prompt=None, mm_processor_kwargs=None, enable_lora=False, max_loras=1, max_lora_rank=16, lora_extra_vocab_size=256, lora_dtype='auto', long_lora_scaling_factors=None, max_cpu_loras=None, fully_sharded_loras=False, enable_prompt_adapter=False, max_prompt_adapters=1, max_prompt_adapter_token=0, device='auto', num_scheduler_steps=1, multi_step_stream_outputs=True, scheduler_delay_factor=0.0, enable_chunked_prefill=True, speculative_model=None, speculative_model_quantization=None, num_speculative_tokens=None, speculative_disable_mqa_scorer=False, speculative_draft_tensor_parallel_size=None, speculative_max_model_len=None, speculative_disable_by_batch_size=None, ngram_prompt_lookup_max=None, ngram_prompt_lookup_min=None, spec_decoding_acceptance_method='rejection_sampler', typical_acceptance_sampler_posterior_threshold=None, typical_acceptance_sampler_posterior_alpha=None, disable_logprobs_during_spec_decoding=None, model_loader_extra_config=None, ignore_patterns=[], preemption_mode=None, served_model_name=['llm'], qlora_adapter_name_or_path=None, otlp_traces_endpoint=None, collect_detailed_traces=None, disable_async_output_proc=False, override_neuron_config=None, scheduling_policy='fcfs', disable_log_requests=True, max_log_len=None, disable_fastapi_docs=False)
|
||||||
|
INFO 07-14 12:25:15 config.py:1670] Downcasting torch.float32 to torch.float16.
|
||||||
|
INFO 07-14 12:25:26 config.py:887] Defaulting to use mp for distributed inference
|
||||||
|
INFO 07-14 12:25:26 config.py:1005] Chunked prefill is enabled with max_num_batched_tokens=8192.
|
||||||
|
WARNING 07-14 12:25:26 config.py:380] To see benefits of async output processing, enable CUDA graph. Since, enforce-eager is enabled, async output processor cannot be used
|
||||||
|
INFO 07-14 12:25:26 llm_engine.py:237] Initializing an LLM engine (v0.6.3) with config: model='/root/public-storage/models/Qwen/Qwen3.6-35B-A3B', speculative_config=None, tokenizer='/root/public-storage/models/Qwen/Qwen3.6-35B-A3B', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, override_neuron_config=None, rope_scaling=None, rope_theta=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.float16, max_seq_len=100000, download_dir=None, load_format=LoadFormat.AUTO, tensor_parallel_size=4, pipeline_parallel_size=1, disable_custom_all_reduce=False, quantization=None, enforce_eager=True, kv_cache_dtype=auto, quantization_param_path=None, device_config=cuda, decoding_config=DecodingConfig(guided_decoding_backend='outlines'), observability_config=ObservabilityConfig(otlp_traces_endpoint=None, collect_model_forward_time=False, collect_model_execute_time=False), seed=0, served_model_name=llm, use_v2_block_manager=True, num_scheduler_steps=1, chunked_prefill_enabled=True multi_step_stream_outputs=True, enable_prefix_caching=True, use_async_output_proc=False, use_cached_outputs=False, mm_processor_kwargs=None)
|
||||||
|
WARNING 07-14 12:25:27 multiproc_gpu_executor.py:53] Reducing Torch parallelism from 64 threads to 1 to avoid unnecessary CPU contention. Set OMP_NUM_THREADS in the external environment to tune this value as needed.
|
||||||
|
INFO 07-14 12:25:27 custom_cache_manager.py:17] Setting Triton cache manager to: vllm.triton_utils.custom_cache_manager:CustomCacheManager
|
||||||
|
INFO 07-14 12:25:27 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
INFO 07-14 12:25:27 selector.py:115] Using XFormers backend.
|
||||||
|
/usr/local/corex/lib/python3/dist-packages/torch/cuda/__init__.py:51: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you.
|
||||||
|
import pynvml # type: ignore[import]
|
||||||
|
/usr/local/corex/lib/python3/dist-packages/torch/cuda/__init__.py:51: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you.
|
||||||
|
import pynvml # type: ignore[import]
|
||||||
|
/usr/local/corex/lib/python3/dist-packages/torch/cuda/__init__.py:51: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you.
|
||||||
|
import pynvml # type: ignore[import]
|
||||||
|
INFO 07-14 12:25:29 importing.py:10] Triton not installed; certain GPU-related functions will not be available.
|
||||||
|
INFO 07-14 12:25:29 importing.py:10] Triton not installed; certain GPU-related functions will not be available.
|
||||||
|
INFO 07-14 12:25:29 importing.py:10] Triton not installed; certain GPU-related functions will not be available.
|
||||||
|
WARNING:tensorflow:Deprecation warnings have been disabled. Set TF_ENABLE_DEPRECATION_WARNINGS=1 to re-enable them.
|
||||||
|
WARNING:tensorflow:Deprecation warnings have been disabled. Set TF_ENABLE_DEPRECATION_WARNINGS=1 to re-enable them.
|
||||||
|
WARNING:tensorflow:Deprecation warnings have been disabled. Set TF_ENABLE_DEPRECATION_WARNINGS=1 to re-enable them.
|
||||||
|
[1;36m(VllmWorkerProcess pid=14382)[0;0m INFO 07-14 12:25:36 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=14382)[0;0m INFO 07-14 12:25:36 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=14380)[0;0m INFO 07-14 12:25:36 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=14380)[0;0m INFO 07-14 12:25:36 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=14381)[0;0m INFO 07-14 12:25:36 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=14381)[0;0m INFO 07-14 12:25:36 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=14382)[0;0m INFO 07-14 12:25:36 multiproc_worker_utils.py:216] Worker ready; awaiting tasks
|
||||||
|
[1;36m(VllmWorkerProcess pid=14380)[0;0m INFO 07-14 12:25:36 multiproc_worker_utils.py:216] Worker ready; awaiting tasks
|
||||||
|
[1;36m(VllmWorkerProcess pid=14381)[0;0m INFO 07-14 12:25:36 multiproc_worker_utils.py:216] Worker ready; awaiting tasks
|
||||||
|
INFO 07-14 12:25:36 shm_broadcast.py:242] vLLM message queue communication handle: Handle(connect_ip='127.0.0.1', local_reader_ranks=[1, 2, 3], buffer=<vllm.distributed.device_communicators.shm_broadcast.ShmRingBuffer object at 0x7f5175238550>, local_subscribe_port=52711, remote_subscribe_port=None)
|
||||||
|
INFO 07-14 12:25:36 model_runner.py:1112] Starting to load model /root/public-storage/models/Qwen/Qwen3.6-35B-A3B...
|
||||||
|
[1;36m(VllmWorkerProcess pid=14380)[0;0m INFO 07-14 12:25:36 model_runner.py:1112] Starting to load model /root/public-storage/models/Qwen/Qwen3.6-35B-A3B...
|
||||||
|
[1;36m(VllmWorkerProcess pid=14381)[0;0m INFO 07-14 12:25:36 model_runner.py:1112] Starting to load model /root/public-storage/models/Qwen/Qwen3.6-35B-A3B...
|
||||||
|
[1;36m(VllmWorkerProcess pid=14382)[0;0m INFO 07-14 12:25:36 model_runner.py:1112] Starting to load model /root/public-storage/models/Qwen/Qwen3.6-35B-A3B...
|
||||||
|
INFO 07-14 12:25:36 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
INFO 07-14 12:25:36 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=14380)[0;0m INFO 07-14 12:25:36 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=14382)[0;0m INFO 07-14 12:25:36 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=14380)[0;0m INFO 07-14 12:25:36 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=14382)[0;0m INFO 07-14 12:25:36 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=14381)[0;0m INFO 07-14 12:25:36 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=14381)[0;0m INFO 07-14 12:25:36 selector.py:115] Using XFormers backend.
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 0% Completed | 0/26 [00:00<?, ?it/s]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 4% Completed | 1/26 [00:01<00:45, 1.81s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 8% Completed | 2/26 [00:02<00:24, 1.02s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 12% Completed | 3/26 [00:04<00:32, 1.42s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 15% Completed | 4/26 [00:06<00:34, 1.59s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 19% Completed | 5/26 [00:06<00:28, 1.37s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 23% Completed | 6/26 [00:09<00:31, 1.59s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 27% Completed | 7/26 [00:09<00:23, 1.22s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 31% Completed | 8/26 [00:11<00:24, 1.39s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 35% Completed | 9/26 [00:12<00:21, 1.25s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 38% Completed | 10/26 [00:14<00:23, 1.47s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 42% Completed | 11/26 [00:14<00:16, 1.11s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 46% Completed | 12/26 [00:16<00:18, 1.31s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 50% Completed | 13/26 [00:18<00:19, 1.48s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 54% Completed | 14/26 [00:20<00:19, 1.66s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 58% Completed | 15/26 [00:20<00:15, 1.39s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 62% Completed | 16/26 [00:23<00:16, 1.68s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 65% Completed | 17/26 [00:23<00:11, 1.31s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 69% Completed | 18/26 [00:25<00:11, 1.46s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 73% Completed | 19/26 [00:27<00:10, 1.52s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 77% Completed | 20/26 [00:29<00:10, 1.78s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 81% Completed | 21/26 [00:30<00:07, 1.55s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 85% Completed | 22/26 [00:31<00:05, 1.30s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 88% Completed | 23/26 [00:33<00:04, 1.61s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 92% Completed | 24/26 [00:35<00:03, 1.70s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 96% Completed | 25/26 [00:37<00:01, 1.76s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 100% Completed | 26/26 [00:37<00:00, 1.38s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 100% Completed | 26/26 [00:37<00:00, 1.46s/it]
|
||||||
|
|
||||||
|
[1;36m(VllmWorkerProcess pid=14380)[0;0m INFO 07-14 12:26:15 model_runner.py:1123] Loading model weights took 16.2303 GB
|
||||||
|
INFO 07-14 12:26:15 model_runner.py:1123] Loading model weights took 16.2303 GB
|
||||||
|
[1;36m(VllmWorkerProcess pid=14381)[0;0m INFO 07-14 12:26:15 model_runner.py:1123] Loading model weights took 16.2303 GB
|
||||||
|
[1;36m(VllmWorkerProcess pid=14382)[0;0m INFO 07-14 12:26:16 model_runner.py:1123] Loading model weights took 16.2303 GB
|
||||||
|
INFO 07-14 12:26:24 distributed_gpu_executor.py:57] # GPU blocks: 21100, # CPU blocks: 6553
|
||||||
|
INFO 07-14 12:26:24 distributed_gpu_executor.py:61] Maximum concurrency for 100000 tokens per request: 3.38x
|
||||||
|
INFO 07-14 12:26:28 serving_chat.py:79] "auto" tool choice has been enabled please note that while the parallel_tool_calls client option is preset for compatibility reasons, it will be ignored.
|
||||||
|
INFO 07-14 12:26:28 serving_chat.py:101] Reasoning parser 'qwen3' enabled.
|
||||||
|
WARNING 07-14 12:26:28 serving_embedding.py:199] embedding_mode is False. Embedding API will not work.
|
||||||
|
INFO 07-14 12:26:28 launcher.py:19] Available routes are:
|
||||||
|
INFO 07-14 12:26:28 launcher.py:27] Route: /openapi.json, Methods: HEAD, GET
|
||||||
|
INFO 07-14 12:26:28 launcher.py:27] Route: /docs, Methods: HEAD, GET
|
||||||
|
INFO 07-14 12:26:28 launcher.py:27] Route: /docs/oauth2-redirect, Methods: HEAD, GET
|
||||||
|
INFO 07-14 12:26:28 launcher.py:27] Route: /redoc, Methods: HEAD, GET
|
||||||
|
INFO 07-14 12:26:28 launcher.py:27] Route: /health, Methods: GET
|
||||||
|
INFO 07-14 12:26:28 launcher.py:27] Route: /tokenize, Methods: POST
|
||||||
|
INFO 07-14 12:26:28 launcher.py:27] Route: /detokenize, Methods: POST
|
||||||
|
INFO 07-14 12:26:28 launcher.py:27] Route: /v1/models, Methods: GET
|
||||||
|
INFO 07-14 12:26:28 launcher.py:27] Route: /version, Methods: GET
|
||||||
|
INFO 07-14 12:26:28 launcher.py:27] Route: /v1/chat/completions, Methods: POST
|
||||||
|
INFO 07-14 12:26:28 launcher.py:27] Route: /v1/completions, Methods: POST
|
||||||
|
INFO 07-14 12:26:28 launcher.py:27] Route: /v1/embeddings, Methods: POST
|
||||||
|
INFO: Started server process [14041]
|
||||||
|
INFO: Waiting for application startup.
|
||||||
|
INFO: Application startup complete.
|
||||||
|
INFO: Uvicorn running on socket ('0.0.0.0', 1111) (Press CTRL+C to quit)
|
||||||
|
INFO 07-14 12:26:39 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:26:39 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:26:49 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:26:49 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:26:59 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:26:59 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:27:09 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:27:09 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:27:19 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:27:19 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:27:29 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:27:29 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:27:39 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:27:39 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO: 127.0.0.1:46086 - "GET /health HTTP/1.1" 200 OK
|
||||||
|
INFO 07-14 12:27:49 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:27:49 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:27:59 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:27:59 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:28:09 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:28:09 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:28:19 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:28:19 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:28:29 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:28:29 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:28:39 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:28:39 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:28:49 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:28:49 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:28:59 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:28:59 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO: 127.0.0.1:40074 - "GET /health HTTP/1.1" 200 OK
|
||||||
|
INFO 07-14 12:29:09 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:29:09 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:29:19 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:29:19 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:29:29 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:29:29 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:29:39 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:29:39 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO: 127.0.0.1:46954 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||||
|
INFO: 127.0.0.1:46970 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||||
|
/usr/local/lib/python3.10/site-packages/pyairports/airports.py:1: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81.
|
||||||
|
from pkg_resources import resource_string
|
||||||
|
INFO 07-14 12:29:49 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 2 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:29:49 metrics.py:361] Prefix cache hit rate: GPU: 50.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:29:54 metrics.py:345] Avg prompt throughput: 15.4 tokens/s, Avg generation throughput: 7.5 tokens/s, Running: 2 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:29:54 metrics.py:361] Prefix cache hit rate: GPU: 50.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:29:59 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 12.0 tokens/s, Running: 2 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:29:59 metrics.py:361] Prefix cache hit rate: GPU: 50.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:30:05 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 11.8 tokens/s, Running: 2 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:30:05 metrics.py:361] Prefix cache hit rate: GPU: 50.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:30:10 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 11.6 tokens/s, Running: 2 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.1%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:30:10 metrics.py:361] Prefix cache hit rate: GPU: 50.00%, CPU: 0.00%
|
||||||
|
INFO: 127.0.0.1:51990 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||||
|
INFO: 127.0.0.1:51998 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||||
|
INFO 07-14 12:30:15 metrics.py:345] Avg prompt throughput: 15.1 tokens/s, Avg generation throughput: 9.3 tokens/s, Running: 2 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:30:15 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:30:20 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 12.0 tokens/s, Running: 2 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:30:20 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:30:25 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 12.1 tokens/s, Running: 2 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:30:25 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:30:30 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 12.2 tokens/s, Running: 2 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:30:30 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:30:39 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 6.6 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:30:39 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:30:49 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:30:49 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:30:59 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:30:59 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:31:09 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:31:09 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:31:19 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:31:19 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:31:29 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:31:29 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:31:39 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:31:39 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:31:49 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:31:49 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:31:59 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:31:59 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:32:09 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:32:09 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:32:19 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:32:19 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:32:29 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:32:29 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:32:39 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:32:39 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:32:49 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:32:49 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:32:59 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:32:59 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:33:09 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:33:09 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:33:19 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:33:19 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:33:29 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:33:29 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:33:39 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:33:39 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:33:49 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:33:49 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:33:59 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:33:59 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:34:09 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:34:09 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:34:19 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:34:19 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:34:29 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:34:29 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:34:39 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:34:39 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:34:49 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:34:49 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:34:59 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:34:59 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:35:09 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:35:09 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:35:19 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:35:19 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:35:29 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:35:29 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:35:39 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:35:39 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:35:49 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:35:49 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:35:59 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:35:59 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:36:09 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:36:09 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:36:19 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:36:19 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:36:29 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:36:29 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:36:39 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:36:39 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:36:49 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:36:49 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:36:59 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:36:59 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:37:09 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:37:09 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:37:19 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:37:19 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:37:29 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:37:29 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:37:39 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:37:39 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:37:49 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:37:49 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:37:59 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:37:59 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:38:09 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:38:09 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:38:19 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:38:19 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:38:29 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:38:29 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:38:39 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:38:39 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:38:49 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:38:49 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:38:59 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:38:59 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 12:39:09 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 12:39:09 metrics.py:361] Prefix cache hit rate: GPU: 75.00%, CPU: 0.00%
|
||||||
@@ -0,0 +1,66 @@
|
|||||||
|
{
|
||||||
|
"created_at": "2026-07-14T08:00:25",
|
||||||
|
"label": "full_parser_short_c1_t256_r3",
|
||||||
|
"url": "http://127.0.0.1:1111",
|
||||||
|
"model": "llm",
|
||||||
|
"prompt_mode": "short",
|
||||||
|
"with_tools": false,
|
||||||
|
"tool_count": 0,
|
||||||
|
"concurrency": 1,
|
||||||
|
"requests": 3,
|
||||||
|
"max_tokens": 256,
|
||||||
|
"wall_sec": 92.0443243663758,
|
||||||
|
"success_rate": 1.0,
|
||||||
|
"ttft_p50_sec": 1.0909903924912214,
|
||||||
|
"ttft_p90_sec": 1.8471774261444807,
|
||||||
|
"output_tps_p10_per_request": 8.736438485385,
|
||||||
|
"output_tps_p50_per_request": 8.738006408510326,
|
||||||
|
"aggregate_output_tps": 8.343806153033741,
|
||||||
|
"prompt_tokens": 117,
|
||||||
|
"cached_tokens": 64,
|
||||||
|
"completion_tokens": 768,
|
||||||
|
"reasoning_tokens": 768,
|
||||||
|
"chars": 2635,
|
||||||
|
"monitor": {
|
||||||
|
"records": 92,
|
||||||
|
"parsed_samples": 0
|
||||||
|
},
|
||||||
|
"results": [
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 31.292513709515333,
|
||||||
|
"ttft_sec": 2.0362241845577955,
|
||||||
|
"completion_tokens": 256,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 0,
|
||||||
|
"reasoning_tokens": 256,
|
||||||
|
"output_tps": 8.750255215433768,
|
||||||
|
"chars": 916,
|
||||||
|
"error": null
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 30.353378538042307,
|
||||||
|
"ttft_sec": 1.0560779832303524,
|
||||||
|
"completion_tokens": 256,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 32,
|
||||||
|
"reasoning_tokens": 256,
|
||||||
|
"output_tps": 8.738006408510326,
|
||||||
|
"chars": 856,
|
||||||
|
"error": null
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 30.394863702356815,
|
||||||
|
"ttft_sec": 1.0909903924912214,
|
||||||
|
"completion_tokens": 256,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 32,
|
||||||
|
"reasoning_tokens": 256,
|
||||||
|
"output_tps": 8.736046504603667,
|
||||||
|
"chars": 863,
|
||||||
|
"error": null
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
@@ -0,0 +1,63 @@
|
|||||||
|
{
|
||||||
|
"created_at": "2026-07-14T08:11:04",
|
||||||
|
"label": "full_parser_tool_c1_t128_r3",
|
||||||
|
"url": "http://127.0.0.1:1111",
|
||||||
|
"model": "llm",
|
||||||
|
"prompt_mode": "tool",
|
||||||
|
"with_tools": true,
|
||||||
|
"tool_count": 29,
|
||||||
|
"concurrency": 1,
|
||||||
|
"requests": 3,
|
||||||
|
"max_tokens": 128,
|
||||||
|
"wall_sec": 52.042631950229406,
|
||||||
|
"success_rate": 1.0,
|
||||||
|
"ttft_p50_sec": 1.0516742002218962,
|
||||||
|
"ttft_p90_sec": 4.98421496860683,
|
||||||
|
"output_tps_p10_per_request": 8.700721568035489,
|
||||||
|
"output_tps_p50_per_request": 8.704939531751927,
|
||||||
|
"aggregate_output_tps": 7.378566102637461,
|
||||||
|
"prompt_tokens": 6639,
|
||||||
|
"cached_tokens": 4416,
|
||||||
|
"completion_tokens": 384,
|
||||||
|
"reasoning_tokens": 384,
|
||||||
|
"chars": 1309,
|
||||||
|
"monitor": null,
|
||||||
|
"results": [
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 20.671645294874907,
|
||||||
|
"ttft_sec": 5.967350160703063,
|
||||||
|
"completion_tokens": 128,
|
||||||
|
"prompt_tokens": 2213,
|
||||||
|
"cached_tokens": 0,
|
||||||
|
"reasoning_tokens": 128,
|
||||||
|
"output_tps": 8.704939531751927,
|
||||||
|
"chars": 445,
|
||||||
|
"error": null
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 15.636568604037166,
|
||||||
|
"ttft_sec": 1.0516742002218962,
|
||||||
|
"completion_tokens": 128,
|
||||||
|
"prompt_tokens": 2213,
|
||||||
|
"cached_tokens": 2208,
|
||||||
|
"reasoning_tokens": 128,
|
||||||
|
"output_tps": 8.776203409914055,
|
||||||
|
"chars": 456,
|
||||||
|
"error": null
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 15.731618992984295,
|
||||||
|
"ttft_sec": 1.018412284553051,
|
||||||
|
"completion_tokens": 128,
|
||||||
|
"prompt_tokens": 2213,
|
||||||
|
"cached_tokens": 2208,
|
||||||
|
"reasoning_tokens": 128,
|
||||||
|
"output_tps": 8.699667077106378,
|
||||||
|
"chars": 408,
|
||||||
|
"error": null
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
@@ -0,0 +1,75 @@
|
|||||||
|
{
|
||||||
|
"created_at": "2026-07-14T08:34:26",
|
||||||
|
"label": "noparser_short_c1_t128_r4",
|
||||||
|
"url": "http://127.0.0.1:1111",
|
||||||
|
"model": "llm",
|
||||||
|
"prompt_mode": "short",
|
||||||
|
"with_tools": false,
|
||||||
|
"tool_count": 0,
|
||||||
|
"concurrency": 1,
|
||||||
|
"requests": 4,
|
||||||
|
"max_tokens": 128,
|
||||||
|
"wall_sec": 64.02519051544368,
|
||||||
|
"success_rate": 1.0,
|
||||||
|
"ttft_p50_sec": 0.7723927367478609,
|
||||||
|
"ttft_p90_sec": 0.7913718132302165,
|
||||||
|
"output_tps_p10_per_request": 8.304385542911493,
|
||||||
|
"output_tps_p50_per_request": 8.42998946937195,
|
||||||
|
"aggregate_output_tps": 7.99685242446095,
|
||||||
|
"prompt_tokens": 156,
|
||||||
|
"cached_tokens": 128,
|
||||||
|
"completion_tokens": 512,
|
||||||
|
"reasoning_tokens": 0,
|
||||||
|
"chars": 1827,
|
||||||
|
"monitor": null,
|
||||||
|
"results": [
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 16.272426838055253,
|
||||||
|
"ttft_sec": 0.7992097493261099,
|
||||||
|
"completion_tokens": 128,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 32,
|
||||||
|
"reasoning_tokens": 0,
|
||||||
|
"output_tps": 8.272358570683828,
|
||||||
|
"chars": 474,
|
||||||
|
"error": null
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 15.848933763802052,
|
||||||
|
"ttft_sec": 0.7730832956731319,
|
||||||
|
"completion_tokens": 128,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 32,
|
||||||
|
"reasoning_tokens": 0,
|
||||||
|
"output_tps": 8.490399945966447,
|
||||||
|
"chars": 449,
|
||||||
|
"error": null
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 16.047778205946088,
|
||||||
|
"ttft_sec": 0.7717021778225899,
|
||||||
|
"completion_tokens": 128,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 32,
|
||||||
|
"reasoning_tokens": 0,
|
||||||
|
"output_tps": 8.379115144776051,
|
||||||
|
"chars": 452,
|
||||||
|
"error": null
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 15.853784639388323,
|
||||||
|
"ttft_sec": 0.7609824072569609,
|
||||||
|
"completion_tokens": 128,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 32,
|
||||||
|
"reasoning_tokens": 0,
|
||||||
|
"output_tps": 8.480863793967849,
|
||||||
|
"chars": 452,
|
||||||
|
"error": null
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
@@ -0,0 +1,63 @@
|
|||||||
|
{
|
||||||
|
"created_at": "2026-07-14T08:23:01",
|
||||||
|
"label": "noparser_short_c1_t256_r3",
|
||||||
|
"url": "http://127.0.0.1:1111",
|
||||||
|
"model": "llm",
|
||||||
|
"prompt_mode": "short",
|
||||||
|
"with_tools": false,
|
||||||
|
"tool_count": 0,
|
||||||
|
"concurrency": 1,
|
||||||
|
"requests": 3,
|
||||||
|
"max_tokens": 256,
|
||||||
|
"wall_sec": 97.52940320037305,
|
||||||
|
"success_rate": 1.0,
|
||||||
|
"ttft_p50_sec": 0.8136040437966585,
|
||||||
|
"ttft_p90_sec": 3.1817516405135393,
|
||||||
|
"output_tps_p10_per_request": 8.295711129981452,
|
||||||
|
"output_tps_p50_per_request": 8.327121469734012,
|
||||||
|
"aggregate_output_tps": 7.874548339254703,
|
||||||
|
"prompt_tokens": 117,
|
||||||
|
"cached_tokens": 64,
|
||||||
|
"completion_tokens": 768,
|
||||||
|
"reasoning_tokens": 0,
|
||||||
|
"chars": 2635,
|
||||||
|
"monitor": null,
|
||||||
|
"results": [
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 34.66234661638737,
|
||||||
|
"ttft_sec": 3.7737885396927595,
|
||||||
|
"completion_tokens": 256,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 0,
|
||||||
|
"reasoning_tokens": 0,
|
||||||
|
"output_tps": 8.287858545043312,
|
||||||
|
"chars": 916,
|
||||||
|
"error": null
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 31.533973263576627,
|
||||||
|
"ttft_sec": 0.7910567671060562,
|
||||||
|
"completion_tokens": 256,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 32,
|
||||||
|
"reasoning_tokens": 0,
|
||||||
|
"output_tps": 8.327121469734012,
|
||||||
|
"chars": 856,
|
||||||
|
"error": null
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 31.330975525081158,
|
||||||
|
"ttft_sec": 0.8136040437966585,
|
||||||
|
"completion_tokens": 256,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 32,
|
||||||
|
"reasoning_tokens": 0,
|
||||||
|
"output_tps": 8.388664802176624,
|
||||||
|
"chars": 863,
|
||||||
|
"error": null
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
@@ -0,0 +1,75 @@
|
|||||||
|
{
|
||||||
|
"created_at": "2026-07-14T08:35:37",
|
||||||
|
"label": "noparser_short_c2_t128_r4",
|
||||||
|
"url": "http://127.0.0.1:1111",
|
||||||
|
"model": "llm",
|
||||||
|
"prompt_mode": "short",
|
||||||
|
"with_tools": false,
|
||||||
|
"tool_count": 0,
|
||||||
|
"concurrency": 2,
|
||||||
|
"requests": 4,
|
||||||
|
"max_tokens": 128,
|
||||||
|
"wall_sec": 70.80830597691238,
|
||||||
|
"success_rate": 1.0,
|
||||||
|
"ttft_p50_sec": 0.9245951194316149,
|
||||||
|
"ttft_p90_sec": 0.9249729935079813,
|
||||||
|
"output_tps_p10_per_request": 3.6832897295816034,
|
||||||
|
"output_tps_p50_per_request": 3.7126825201316764,
|
||||||
|
"aggregate_output_tps": 7.230790130284175,
|
||||||
|
"prompt_tokens": 156,
|
||||||
|
"cached_tokens": 128,
|
||||||
|
"completion_tokens": 512,
|
||||||
|
"reasoning_tokens": 0,
|
||||||
|
"chars": 1816,
|
||||||
|
"monitor": null,
|
||||||
|
"results": [
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 35.12894960306585,
|
||||||
|
"ttft_sec": 0.9234285913407803,
|
||||||
|
"completion_tokens": 128,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 32,
|
||||||
|
"reasoning_tokens": 0,
|
||||||
|
"output_tps": 3.742085961974495,
|
||||||
|
"chars": 454,
|
||||||
|
"error": null
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 35.12996072135866,
|
||||||
|
"ttft_sec": 0.9242945611476898,
|
||||||
|
"completion_tokens": 128,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 32,
|
||||||
|
"reasoning_tokens": 0,
|
||||||
|
"output_tps": 3.7420700827891884,
|
||||||
|
"chars": 454,
|
||||||
|
"error": null
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 35.67656988278031,
|
||||||
|
"ttft_sec": 0.925006128847599,
|
||||||
|
"completion_tokens": 128,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 32,
|
||||||
|
"reasoning_tokens": 0,
|
||||||
|
"output_tps": 3.683287489056221,
|
||||||
|
"chars": 454,
|
||||||
|
"error": null
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 35.67638896778226,
|
||||||
|
"ttft_sec": 0.9248956777155399,
|
||||||
|
"completion_tokens": 128,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 32,
|
||||||
|
"reasoning_tokens": 0,
|
||||||
|
"output_tps": 3.683294957474164,
|
||||||
|
"chars": 454,
|
||||||
|
"error": null
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
@@ -0,0 +1,84 @@
|
|||||||
|
{
|
||||||
|
"created_at": "2026-07-14T08:56:17",
|
||||||
|
"label": "noparser_short_c2_t128_r4_monitor",
|
||||||
|
"url": "http://127.0.0.1:1111",
|
||||||
|
"model": "llm",
|
||||||
|
"prompt_mode": "short",
|
||||||
|
"with_tools": false,
|
||||||
|
"tool_count": 0,
|
||||||
|
"concurrency": 2,
|
||||||
|
"requests": 4,
|
||||||
|
"max_tokens": 128,
|
||||||
|
"wall_sec": 69.160331716761,
|
||||||
|
"success_rate": 1.0,
|
||||||
|
"ttft_p50_sec": 0.9177342765033245,
|
||||||
|
"ttft_p90_sec": 0.9243953077122569,
|
||||||
|
"output_tps_p10_per_request": 3.7938414669033445,
|
||||||
|
"output_tps_p50_per_request": 3.8028299895382536,
|
||||||
|
"aggregate_output_tps": 7.4030876846693445,
|
||||||
|
"prompt_tokens": 156,
|
||||||
|
"cached_tokens": 128,
|
||||||
|
"completion_tokens": 512,
|
||||||
|
"reasoning_tokens": 0,
|
||||||
|
"chars": 1816,
|
||||||
|
"monitor": {
|
||||||
|
"records": 52,
|
||||||
|
"parsed_samples": 208,
|
||||||
|
"avg_gpu_util_pct": 28.91346153846154,
|
||||||
|
"max_gpu_util_pct": 100,
|
||||||
|
"avg_mem_mib": 30477.0,
|
||||||
|
"max_mem_mib": 30555,
|
||||||
|
"avg_power_w": 49.67307692307692,
|
||||||
|
"max_power_w": 51
|
||||||
|
},
|
||||||
|
"results": [
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 34.663240656256676,
|
||||||
|
"ttft_sec": 0.9245693255215883,
|
||||||
|
"completion_tokens": 128,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 32,
|
||||||
|
"reasoning_tokens": 0,
|
||||||
|
"output_tps": 3.7938660578905252,
|
||||||
|
"chars": 454,
|
||||||
|
"error": null
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 34.66297300904989,
|
||||||
|
"ttft_sec": 0.9239892661571503,
|
||||||
|
"completion_tokens": 128,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 32,
|
||||||
|
"reasoning_tokens": 0,
|
||||||
|
"output_tps": 3.793830927908839,
|
||||||
|
"chars": 454,
|
||||||
|
"error": null
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 34.49104499258101,
|
||||||
|
"ttft_sec": 0.9110554531216621,
|
||||||
|
"completion_tokens": 128,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 32,
|
||||||
|
"reasoning_tokens": 0,
|
||||||
|
"output_tps": 3.811793921185982,
|
||||||
|
"chars": 454,
|
||||||
|
"error": null
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 34.491438234224916,
|
||||||
|
"ttft_sec": 0.9114792868494987,
|
||||||
|
"completion_tokens": 128,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 32,
|
||||||
|
"reasoning_tokens": 0,
|
||||||
|
"output_tps": 3.811797393814395,
|
||||||
|
"chars": 454,
|
||||||
|
"error": null
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
@@ -0,0 +1,75 @@
|
|||||||
|
{
|
||||||
|
"created_at": "2026-07-14T08:37:18",
|
||||||
|
"label": "noparser_short_c4_t128_r4",
|
||||||
|
"url": "http://127.0.0.1:1111",
|
||||||
|
"model": "llm",
|
||||||
|
"prompt_mode": "short",
|
||||||
|
"with_tools": false,
|
||||||
|
"tool_count": 0,
|
||||||
|
"concurrency": 4,
|
||||||
|
"requests": 4,
|
||||||
|
"max_tokens": 128,
|
||||||
|
"wall_sec": 100.87516433186829,
|
||||||
|
"success_rate": 1.0,
|
||||||
|
"ttft_p50_sec": 26.261823972687125,
|
||||||
|
"ttft_p90_sec": 51.59956250209361,
|
||||||
|
"output_tps_p10_per_request": 2.5736346925937763,
|
||||||
|
"output_tps_p50_per_request": 2.585721667979742,
|
||||||
|
"aggregate_output_tps": 5.075580331305096,
|
||||||
|
"prompt_tokens": 156,
|
||||||
|
"cached_tokens": 128,
|
||||||
|
"completion_tokens": 512,
|
||||||
|
"reasoning_tokens": 0,
|
||||||
|
"chars": 1846,
|
||||||
|
"monitor": null,
|
||||||
|
"results": [
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 50.6585742700845,
|
||||||
|
"ttft_sec": 0.9234585296362638,
|
||||||
|
"completion_tokens": 128,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 32,
|
||||||
|
"reasoning_tokens": 0,
|
||||||
|
"output_tps": 2.573634304341249,
|
||||||
|
"chars": 454,
|
||||||
|
"error": null
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 50.65938341990113,
|
||||||
|
"ttft_sec": 0.9242926891893148,
|
||||||
|
"completion_tokens": 128,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 32,
|
||||||
|
"reasoning_tokens": 0,
|
||||||
|
"output_tps": 2.5736355985163404,
|
||||||
|
"chars": 486,
|
||||||
|
"error": null
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 100.87196587957442,
|
||||||
|
"ttft_sec": 51.59965132176876,
|
||||||
|
"completion_tokens": 128,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 32,
|
||||||
|
"reasoning_tokens": 0,
|
||||||
|
"output_tps": 2.597807737443144,
|
||||||
|
"chars": 454,
|
||||||
|
"error": null
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 100.8716309145093,
|
||||||
|
"ttft_sec": 51.599355256184936,
|
||||||
|
"completion_tokens": 128,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 32,
|
||||||
|
"reasoning_tokens": 0,
|
||||||
|
"output_tps": 2.597809788360666,
|
||||||
|
"chars": 452,
|
||||||
|
"error": null
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,63 @@
|
|||||||
|
{
|
||||||
|
"created_at": "2026-07-14T11:00:49",
|
||||||
|
"label": "eager_custom_ar_short_c1_t256_r3",
|
||||||
|
"url": "http://127.0.0.1:1111",
|
||||||
|
"model": "llm",
|
||||||
|
"prompt_mode": "short",
|
||||||
|
"with_tools": false,
|
||||||
|
"tool_count": 0,
|
||||||
|
"concurrency": 1,
|
||||||
|
"requests": 3,
|
||||||
|
"max_tokens": 256,
|
||||||
|
"wall_sec": 96.7806523796171,
|
||||||
|
"success_rate": 1.0,
|
||||||
|
"ttft_p50_sec": 1.0768930949270725,
|
||||||
|
"ttft_p90_sec": 3.4729231126606463,
|
||||||
|
"output_tps_p10_per_request": 8.472860425455258,
|
||||||
|
"output_tps_p50_per_request": 8.483117808158335,
|
||||||
|
"aggregate_output_tps": 7.935470376739762,
|
||||||
|
"prompt_tokens": 117,
|
||||||
|
"cached_tokens": 64,
|
||||||
|
"completion_tokens": 768,
|
||||||
|
"reasoning_tokens": 768,
|
||||||
|
"chars": 2635,
|
||||||
|
"monitor": null,
|
||||||
|
"results": [
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 34.2951952572912,
|
||||||
|
"ttft_sec": 4.07193061709404,
|
||||||
|
"completion_tokens": 256,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 0,
|
||||||
|
"reasoning_tokens": 256,
|
||||||
|
"output_tps": 8.470296079779487,
|
||||||
|
"chars": 916,
|
||||||
|
"error": null
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 31.25290015526116,
|
||||||
|
"ttft_sec": 1.0768930949270725,
|
||||||
|
"completion_tokens": 256,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 32,
|
||||||
|
"reasoning_tokens": 256,
|
||||||
|
"output_tps": 8.483561111586171,
|
||||||
|
"chars": 856,
|
||||||
|
"error": null
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 31.230418637394905,
|
||||||
|
"ttft_sec": 1.05283466540277,
|
||||||
|
"completion_tokens": 256,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 32,
|
||||||
|
"reasoning_tokens": 256,
|
||||||
|
"output_tps": 8.483117808158335,
|
||||||
|
"chars": 863,
|
||||||
|
"error": null
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
@@ -0,0 +1,75 @@
|
|||||||
|
{
|
||||||
|
"created_at": "2026-07-14T11:01:53",
|
||||||
|
"label": "eager_custom_ar_short_c2_t128_r4",
|
||||||
|
"url": "http://127.0.0.1:1111",
|
||||||
|
"model": "llm",
|
||||||
|
"prompt_mode": "short",
|
||||||
|
"with_tools": false,
|
||||||
|
"tool_count": 0,
|
||||||
|
"concurrency": 2,
|
||||||
|
"requests": 4,
|
||||||
|
"max_tokens": 128,
|
||||||
|
"wall_sec": 63.41567398421466,
|
||||||
|
"success_rate": 1.0,
|
||||||
|
"ttft_p50_sec": 1.449904115870595,
|
||||||
|
"ttft_p90_sec": 1.4555151607841255,
|
||||||
|
"output_tps_p10_per_request": 4.218399789762928,
|
||||||
|
"output_tps_p50_per_request": 4.2304733122443885,
|
||||||
|
"aggregate_output_tps": 8.073713765581775,
|
||||||
|
"prompt_tokens": 156,
|
||||||
|
"cached_tokens": 128,
|
||||||
|
"completion_tokens": 512,
|
||||||
|
"reasoning_tokens": 512,
|
||||||
|
"chars": 1816,
|
||||||
|
"monitor": null,
|
||||||
|
"results": [
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 31.78747241385281,
|
||||||
|
"ttft_sec": 1.4443142116069794,
|
||||||
|
"completion_tokens": 128,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 32,
|
||||||
|
"reasoning_tokens": 128,
|
||||||
|
"output_tps": 4.21841388911607,
|
||||||
|
"chars": 454,
|
||||||
|
"error": null
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 31.787043346092105,
|
||||||
|
"ttft_sec": 1.4437402617186308,
|
||||||
|
"completion_tokens": 128,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 32,
|
||||||
|
"reasoning_tokens": 128,
|
||||||
|
"output_tps": 4.2183937471830095,
|
||||||
|
"chars": 454,
|
||||||
|
"error": null
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 31.626181228086352,
|
||||||
|
"ttft_sec": 1.4555242210626602,
|
||||||
|
"completion_tokens": 128,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 32,
|
||||||
|
"reasoning_tokens": 128,
|
||||||
|
"output_tps": 4.242532735372708,
|
||||||
|
"chars": 454,
|
||||||
|
"error": null
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ok": true,
|
||||||
|
"elapsed_sec": 31.625997802242637,
|
||||||
|
"ttft_sec": 1.4554940201342106,
|
||||||
|
"completion_tokens": 128,
|
||||||
|
"prompt_tokens": 39,
|
||||||
|
"cached_tokens": 32,
|
||||||
|
"reasoning_tokens": 128,
|
||||||
|
"output_tps": 4.24255428163934,
|
||||||
|
"chars": 454,
|
||||||
|
"error": null
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
@@ -0,0 +1,398 @@
|
|||||||
|
/usr/local/corex/lib/python3/dist-packages/torch/cuda/__init__.py:51: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you.
|
||||||
|
import pynvml # type: ignore[import]
|
||||||
|
INFO 07-14 10:55:02 importing.py:10] Triton not installed; certain GPU-related functions will not be available.
|
||||||
|
2026-07-14 10:55:03.598198: I tensorflow/core/util/port.cc:110] oneDNN custom operations are on. You may see slightly different numerical results due to floating-point round-off errors from different computation orders. To turn them off, set the environment variable `TF_ENABLE_ONEDNN_OPTS=0`.
|
||||||
|
2026-07-14 10:55:03.649443: I tensorflow/core/platform/cpu_feature_guard.cc:182] This TensorFlow binary is optimized to use available CPU instructions in performance-critical operations.
|
||||||
|
To enable the following instructions: SSE3 SSE4.1 SSE4.2 AVX AVX2 AVX512F AVX512_VNNI AVX512_BF16 AVX_VNNI AMX_TILE AMX_INT8 AMX_BF16 FMA, in other operations, rebuild TensorFlow with the appropriate compiler flags.
|
||||||
|
WARNING:tensorflow:Deprecation warnings have been disabled. Set TF_ENABLE_DEPRECATION_WARNINGS=1 to re-enable them.
|
||||||
|
INFO 07-14 10:55:09 api_server.py:530] vLLM API server version 0.6.3
|
||||||
|
INFO 07-14 10:55:09 api_server.py:531] args: Namespace(host='0.0.0.0', port=1111, uvicorn_log_level='info', allow_credentials=False, allowed_origins=['*'], allowed_methods=['*'], allowed_headers=['*'], api_key=None, lora_modules=None, prompt_adapters=None, chat_template=None, response_role='assistant', ssl_keyfile=None, ssl_certfile=None, ssl_ca_certs=None, ssl_cert_reqs=0, root_path=None, middleware=[], return_tokens_as_token_ids=False, disable_frontend_multiprocessing=True, enable_auto_tool_choice=True, tool_call_parser='qwen3_coder', tool_parser_plugin='', reasoning_parser='qwen3', model='/root/public-storage/models/Qwen/Qwen3.6-35B-A3B', tokenizer=None, skip_tokenizer_init=False, revision=None, code_revision=None, tokenizer_revision=None, tokenizer_mode='auto', trust_remote_code=True, download_dir=None, load_format='auto', config_format='auto', dtype='auto', kv_cache_dtype='auto', quantization_param_path=None, max_model_len=100000, guided_decoding_backend='outlines', distributed_executor_backend=None, worker_use_ray=False, pipeline_parallel_size=1, tensor_parallel_size=4, max_parallel_loading_workers=None, ray_workers_use_nsight=False, block_size=16, enable_prefix_caching=True, disable_sliding_window=False, use_v2_block_manager=True, num_lookahead_slots=0, seed=0, swap_space=4, cpu_offload_gb=0, gpu_memory_utilization=0.95, num_gpu_blocks_override=None, max_num_batched_tokens=8192, max_num_seqs=2, max_logprobs=20, disable_log_stats=False, quantization=None, rope_scaling=None, rope_theta=None, enforce_eager=True, max_context_len_to_capture=None, max_seq_len_to_capture=32768, disable_custom_all_reduce=False, tokenizer_pool_size=0, tokenizer_pool_type='ray', tokenizer_pool_extra_config=None, limit_mm_per_prompt=None, mm_processor_kwargs=None, enable_lora=False, max_loras=1, max_lora_rank=16, lora_extra_vocab_size=256, lora_dtype='auto', long_lora_scaling_factors=None, max_cpu_loras=None, fully_sharded_loras=False, enable_prompt_adapter=False, max_prompt_adapters=1, max_prompt_adapter_token=0, device='auto', num_scheduler_steps=1, multi_step_stream_outputs=True, scheduler_delay_factor=0.0, enable_chunked_prefill=True, speculative_model=None, speculative_model_quantization=None, num_speculative_tokens=None, speculative_disable_mqa_scorer=False, speculative_draft_tensor_parallel_size=None, speculative_max_model_len=None, speculative_disable_by_batch_size=None, ngram_prompt_lookup_max=None, ngram_prompt_lookup_min=None, spec_decoding_acceptance_method='rejection_sampler', typical_acceptance_sampler_posterior_threshold=None, typical_acceptance_sampler_posterior_alpha=None, disable_logprobs_during_spec_decoding=None, model_loader_extra_config=None, ignore_patterns=[], preemption_mode=None, served_model_name=['llm'], qlora_adapter_name_or_path=None, otlp_traces_endpoint=None, collect_detailed_traces=None, disable_async_output_proc=False, override_neuron_config=None, scheduling_policy='fcfs', disable_log_requests=True, max_log_len=None, disable_fastapi_docs=False)
|
||||||
|
INFO 07-14 10:55:09 config.py:1670] Downcasting torch.float32 to torch.float16.
|
||||||
|
INFO 07-14 10:55:20 config.py:887] Defaulting to use mp for distributed inference
|
||||||
|
INFO 07-14 10:55:20 config.py:1005] Chunked prefill is enabled with max_num_batched_tokens=8192.
|
||||||
|
WARNING 07-14 10:55:20 config.py:380] To see benefits of async output processing, enable CUDA graph. Since, enforce-eager is enabled, async output processor cannot be used
|
||||||
|
INFO 07-14 10:55:20 llm_engine.py:237] Initializing an LLM engine (v0.6.3) with config: model='/root/public-storage/models/Qwen/Qwen3.6-35B-A3B', speculative_config=None, tokenizer='/root/public-storage/models/Qwen/Qwen3.6-35B-A3B', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, override_neuron_config=None, rope_scaling=None, rope_theta=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.float16, max_seq_len=100000, download_dir=None, load_format=LoadFormat.AUTO, tensor_parallel_size=4, pipeline_parallel_size=1, disable_custom_all_reduce=False, quantization=None, enforce_eager=True, kv_cache_dtype=auto, quantization_param_path=None, device_config=cuda, decoding_config=DecodingConfig(guided_decoding_backend='outlines'), observability_config=ObservabilityConfig(otlp_traces_endpoint=None, collect_model_forward_time=False, collect_model_execute_time=False), seed=0, served_model_name=llm, use_v2_block_manager=True, num_scheduler_steps=1, chunked_prefill_enabled=True multi_step_stream_outputs=True, enable_prefix_caching=True, use_async_output_proc=False, use_cached_outputs=False, mm_processor_kwargs=None)
|
||||||
|
WARNING 07-14 10:55:20 multiproc_gpu_executor.py:53] Reducing Torch parallelism from 64 threads to 1 to avoid unnecessary CPU contention. Set OMP_NUM_THREADS in the external environment to tune this value as needed.
|
||||||
|
INFO 07-14 10:55:20 custom_cache_manager.py:17] Setting Triton cache manager to: vllm.triton_utils.custom_cache_manager:CustomCacheManager
|
||||||
|
INFO 07-14 10:55:20 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
INFO 07-14 10:55:20 selector.py:115] Using XFormers backend.
|
||||||
|
/usr/local/corex/lib/python3/dist-packages/torch/cuda/__init__.py:51: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you.
|
||||||
|
import pynvml # type: ignore[import]
|
||||||
|
/usr/local/corex/lib/python3/dist-packages/torch/cuda/__init__.py:51: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you.
|
||||||
|
import pynvml # type: ignore[import]
|
||||||
|
/usr/local/corex/lib/python3/dist-packages/torch/cuda/__init__.py:51: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you.
|
||||||
|
import pynvml # type: ignore[import]
|
||||||
|
INFO 07-14 10:55:22 importing.py:10] Triton not installed; certain GPU-related functions will not be available.
|
||||||
|
INFO 07-14 10:55:22 importing.py:10] Triton not installed; certain GPU-related functions will not be available.
|
||||||
|
INFO 07-14 10:55:22 importing.py:10] Triton not installed; certain GPU-related functions will not be available.
|
||||||
|
WARNING:tensorflow:Deprecation warnings have been disabled. Set TF_ENABLE_DEPRECATION_WARNINGS=1 to re-enable them.
|
||||||
|
WARNING:tensorflow:Deprecation warnings have been disabled. Set TF_ENABLE_DEPRECATION_WARNINGS=1 to re-enable them.
|
||||||
|
WARNING:tensorflow:Deprecation warnings have been disabled. Set TF_ENABLE_DEPRECATION_WARNINGS=1 to re-enable them.
|
||||||
|
[1;36m(VllmWorkerProcess pid=9684)[0;0m INFO 07-14 10:55:29 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=9684)[0;0m INFO 07-14 10:55:29 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=9684)[0;0m INFO 07-14 10:55:29 multiproc_worker_utils.py:216] Worker ready; awaiting tasks
|
||||||
|
[1;36m(VllmWorkerProcess pid=9683)[0;0m INFO 07-14 10:55:29 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=9683)[0;0m INFO 07-14 10:55:29 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=9683)[0;0m INFO 07-14 10:55:29 multiproc_worker_utils.py:216] Worker ready; awaiting tasks
|
||||||
|
[1;36m(VllmWorkerProcess pid=9685)[0;0m INFO 07-14 10:55:30 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=9685)[0;0m INFO 07-14 10:55:30 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=9685)[0;0m INFO 07-14 10:55:30 multiproc_worker_utils.py:216] Worker ready; awaiting tasks
|
||||||
|
INFO 07-14 10:55:30 shm_broadcast.py:242] vLLM message queue communication handle: Handle(connect_ip='127.0.0.1', local_reader_ranks=[1, 2, 3], buffer=<vllm.distributed.device_communicators.shm_broadcast.ShmRingBuffer object at 0x7f63d51f5a50>, local_subscribe_port=46879, remote_subscribe_port=None)
|
||||||
|
INFO 07-14 10:55:30 model_runner.py:1065] Starting to load model /root/public-storage/models/Qwen/Qwen3.6-35B-A3B...
|
||||||
|
[1;36m(VllmWorkerProcess pid=9684)[0;0m INFO 07-14 10:55:30 model_runner.py:1065] Starting to load model /root/public-storage/models/Qwen/Qwen3.6-35B-A3B...
|
||||||
|
[1;36m(VllmWorkerProcess pid=9685)[0;0m INFO 07-14 10:55:30 model_runner.py:1065] Starting to load model /root/public-storage/models/Qwen/Qwen3.6-35B-A3B...
|
||||||
|
[1;36m(VllmWorkerProcess pid=9683)[0;0m INFO 07-14 10:55:30 model_runner.py:1065] Starting to load model /root/public-storage/models/Qwen/Qwen3.6-35B-A3B...
|
||||||
|
INFO 07-14 10:55:30 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
INFO 07-14 10:55:30 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=9684)[0;0m INFO 07-14 10:55:30 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=9684)[0;0m INFO 07-14 10:55:30 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=9683)[0;0m INFO 07-14 10:55:30 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=9683)[0;0m INFO 07-14 10:55:30 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=9685)[0;0m INFO 07-14 10:55:30 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=9685)[0;0m INFO 07-14 10:55:30 selector.py:115] Using XFormers backend.
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 0% Completed | 0/26 [00:00<?, ?it/s]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 4% Completed | 1/26 [00:01<00:44, 1.77s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 8% Completed | 2/26 [00:02<00:25, 1.05s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 12% Completed | 3/26 [00:04<00:31, 1.39s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 15% Completed | 4/26 [00:06<00:35, 1.61s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 19% Completed | 5/26 [00:07<00:29, 1.38s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 23% Completed | 6/26 [00:08<00:31, 1.58s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 27% Completed | 7/26 [00:09<00:23, 1.21s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 31% Completed | 8/26 [00:11<00:24, 1.39s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 35% Completed | 9/26 [00:12<00:21, 1.25s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 38% Completed | 10/26 [00:14<00:23, 1.47s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 42% Completed | 11/26 [00:14<00:16, 1.12s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 46% Completed | 12/26 [00:16<00:18, 1.34s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 50% Completed | 13/26 [00:18<00:19, 1.50s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 54% Completed | 14/26 [00:20<00:20, 1.69s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 58% Completed | 15/26 [00:21<00:15, 1.43s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 62% Completed | 16/26 [00:23<00:17, 1.73s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 65% Completed | 17/26 [00:24<00:12, 1.38s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 69% Completed | 18/26 [00:26<00:12, 1.55s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 73% Completed | 19/26 [00:27<00:11, 1.60s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 77% Completed | 20/26 [00:30<00:11, 1.87s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 81% Completed | 21/26 [00:31<00:08, 1.62s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 85% Completed | 22/26 [00:32<00:05, 1.35s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 88% Completed | 23/26 [00:34<00:04, 1.65s/it]
|
||||||
|
[1;36m(VllmWorkerProcess pid=9685)[0;0m INFO 07-14 10:56:06 model_runner.py:1076] Loading model weights took 16.2303 GB
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 92% Completed | 24/26 [00:36<00:03, 1.76s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 96% Completed | 25/26 [00:38<00:01, 1.82s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 100% Completed | 26/26 [00:38<00:00, 1.43s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 100% Completed | 26/26 [00:38<00:00, 1.50s/it]
|
||||||
|
|
||||||
|
[1;36m(VllmWorkerProcess pid=9684)[0;0m INFO 07-14 10:56:09 model_runner.py:1076] Loading model weights took 16.2303 GB
|
||||||
|
[1;36m(VllmWorkerProcess pid=9683)[0;0m INFO 07-14 10:56:09 model_runner.py:1076] Loading model weights took 16.2303 GB
|
||||||
|
INFO 07-14 10:56:09 model_runner.py:1076] Loading model weights took 16.2303 GB
|
||||||
|
INFO 07-14 10:56:17 distributed_gpu_executor.py:57] # GPU blocks: 21100, # CPU blocks: 6553
|
||||||
|
INFO 07-14 10:56:17 distributed_gpu_executor.py:61] Maximum concurrency for 100000 tokens per request: 3.38x
|
||||||
|
INFO 07-14 10:56:21 serving_chat.py:79] "auto" tool choice has been enabled please note that while the parallel_tool_calls client option is preset for compatibility reasons, it will be ignored.
|
||||||
|
INFO 07-14 10:56:21 serving_chat.py:101] Reasoning parser 'qwen3' enabled.
|
||||||
|
WARNING 07-14 10:56:21 serving_embedding.py:199] embedding_mode is False. Embedding API will not work.
|
||||||
|
INFO 07-14 10:56:21 launcher.py:19] Available routes are:
|
||||||
|
INFO 07-14 10:56:21 launcher.py:27] Route: /openapi.json, Methods: HEAD, GET
|
||||||
|
INFO 07-14 10:56:21 launcher.py:27] Route: /docs, Methods: HEAD, GET
|
||||||
|
INFO 07-14 10:56:21 launcher.py:27] Route: /docs/oauth2-redirect, Methods: HEAD, GET
|
||||||
|
INFO 07-14 10:56:21 launcher.py:27] Route: /redoc, Methods: HEAD, GET
|
||||||
|
INFO 07-14 10:56:21 launcher.py:27] Route: /health, Methods: GET
|
||||||
|
INFO 07-14 10:56:21 launcher.py:27] Route: /tokenize, Methods: POST
|
||||||
|
INFO 07-14 10:56:21 launcher.py:27] Route: /detokenize, Methods: POST
|
||||||
|
INFO 07-14 10:56:21 launcher.py:27] Route: /v1/models, Methods: GET
|
||||||
|
INFO 07-14 10:56:21 launcher.py:27] Route: /version, Methods: GET
|
||||||
|
INFO 07-14 10:56:21 launcher.py:27] Route: /v1/chat/completions, Methods: POST
|
||||||
|
INFO 07-14 10:56:21 launcher.py:27] Route: /v1/completions, Methods: POST
|
||||||
|
INFO 07-14 10:56:21 launcher.py:27] Route: /v1/embeddings, Methods: POST
|
||||||
|
INFO: Started server process [9342]
|
||||||
|
INFO: Waiting for application startup.
|
||||||
|
INFO: Application startup complete.
|
||||||
|
INFO: Uvicorn running on socket ('0.0.0.0', 1111) (Press CTRL+C to quit)
|
||||||
|
INFO 07-14 10:56:31 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 10:56:31 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO: 127.0.0.1:44316 - "GET /health HTTP/1.1" 200 OK
|
||||||
|
INFO 07-14 10:56:41 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 10:56:41 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 10:56:51 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 10:56:51 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 10:57:01 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 10:57:01 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 10:57:11 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 10:57:11 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 10:57:21 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 10:57:21 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 10:57:31 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 10:57:31 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 10:57:41 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 10:57:41 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO: 127.0.0.1:49340 - "GET /health HTTP/1.1" 200 OK
|
||||||
|
INFO 07-14 10:57:51 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 10:57:51 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 10:58:01 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 10:58:01 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 10:58:11 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 10:58:11 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 10:58:21 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 10:58:21 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 10:58:31 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 10:58:31 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 10:58:41 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 10:58:41 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 10:58:51 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 10:58:51 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 10:59:01 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 10:59:01 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 10:59:11 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 10:59:11 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO: 127.0.0.1:35214 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||||
|
/usr/local/lib/python3.10/site-packages/pyairports/airports.py:1: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81.
|
||||||
|
from pkg_resources import resource_string
|
||||||
|
INFO 07-14 10:59:16 metrics.py:345] Avg prompt throughput: 7.7 tokens/s, Avg generation throughput: 0.2 tokens/s, Running: 1 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 10:59:16 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 10:59:21 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 8.4 tokens/s, Running: 1 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 10:59:21 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 10:59:26 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 8.5 tokens/s, Running: 1 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 10:59:26 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 10:59:31 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 8.5 tokens/s, Running: 1 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.1%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 10:59:31 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 10:59:36 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 8.4 tokens/s, Running: 1 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.1%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 10:59:36 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 10:59:42 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 8.4 tokens/s, Running: 1 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.1%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 10:59:42 metrics.py:361] Prefix cache hit rate: GPU: 0.00%, CPU: 0.00%
|
||||||
|
INFO: 127.0.0.1:41594 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||||
|
INFO 07-14 10:59:48 metrics.py:345] Avg prompt throughput: 6.5 tokens/s, Avg generation throughput: 7.1 tokens/s, Running: 1 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 10:59:48 metrics.py:361] Prefix cache hit rate: GPU: 50.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 10:59:53 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 8.4 tokens/s, Running: 1 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 10:59:53 metrics.py:361] Prefix cache hit rate: GPU: 50.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 10:59:58 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 8.5 tokens/s, Running: 1 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 10:59:58 metrics.py:361] Prefix cache hit rate: GPU: 50.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:00:03 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 8.4 tokens/s, Running: 1 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.1%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:00:03 metrics.py:361] Prefix cache hit rate: GPU: 50.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:00:08 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 8.4 tokens/s, Running: 1 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.1%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:00:08 metrics.py:361] Prefix cache hit rate: GPU: 50.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:00:13 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 8.4 tokens/s, Running: 1 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.1%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:00:13 metrics.py:361] Prefix cache hit rate: GPU: 50.00%, CPU: 0.00%
|
||||||
|
INFO: 127.0.0.1:41688 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||||
|
INFO 07-14 11:00:19 metrics.py:345] Avg prompt throughput: 6.7 tokens/s, Avg generation throughput: 7.1 tokens/s, Running: 1 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:00:19 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:00:24 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 8.4 tokens/s, Running: 1 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:00:24 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:00:29 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 8.5 tokens/s, Running: 1 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:00:29 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:00:34 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 8.5 tokens/s, Running: 1 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.1%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:00:34 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:00:39 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 8.5 tokens/s, Running: 1 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.1%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:00:39 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:00:44 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 8.5 tokens/s, Running: 1 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.1%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:00:44 metrics.py:361] Prefix cache hit rate: GPU: 66.67%, CPU: 0.00%
|
||||||
|
INFO: 127.0.0.1:39344 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||||
|
INFO: 127.0.0.1:39346 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||||
|
INFO 07-14 11:00:51 metrics.py:345] Avg prompt throughput: 12.2 tokens/s, Avg generation throughput: 6.7 tokens/s, Running: 2 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:00:51 metrics.py:361] Prefix cache hit rate: GPU: 80.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:00:56 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 8.5 tokens/s, Running: 2 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:00:56 metrics.py:361] Prefix cache hit rate: GPU: 80.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:01:01 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 8.5 tokens/s, Running: 2 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:01:01 metrics.py:361] Prefix cache hit rate: GPU: 80.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:01:06 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 8.2 tokens/s, Running: 2 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:01:06 metrics.py:361] Prefix cache hit rate: GPU: 80.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:01:11 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 8.6 tokens/s, Running: 2 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:01:11 metrics.py:361] Prefix cache hit rate: GPU: 80.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:01:16 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 8.2 tokens/s, Running: 2 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.1%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:01:16 metrics.py:361] Prefix cache hit rate: GPU: 80.00%, CPU: 0.00%
|
||||||
|
INFO: 127.0.0.1:42354 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||||
|
INFO: 127.0.0.1:42356 - "POST /v1/chat/completions HTTP/1.1" 200 OK
|
||||||
|
INFO 07-14 11:01:21 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 7.4 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:01:21 metrics.py:361] Prefix cache hit rate: GPU: 80.00%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:01:27 metrics.py:345] Avg prompt throughput: 15.4 tokens/s, Avg generation throughput: 7.5 tokens/s, Running: 2 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:01:27 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:01:32 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 8.5 tokens/s, Running: 2 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:01:32 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:01:37 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 8.3 tokens/s, Running: 2 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:01:37 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:01:42 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 8.4 tokens/s, Running: 2 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:01:42 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:01:47 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 8.4 tokens/s, Running: 2 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:01:47 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:01:52 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 8.4 tokens/s, Running: 2 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.1%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:01:52 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:02:01 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.2 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:02:01 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:02:11 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:02:11 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:02:21 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:02:21 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:02:31 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:02:31 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:02:41 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:02:41 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:02:51 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:02:51 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:03:01 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:03:01 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:03:11 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:03:11 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:03:21 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:03:21 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:03:31 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:03:31 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:03:41 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:03:41 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:03:51 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:03:51 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:04:01 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:04:01 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:04:11 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:04:11 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:04:21 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:04:21 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:04:31 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:04:31 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:04:41 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:04:41 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:04:51 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:04:51 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:05:01 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:05:01 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:05:11 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:05:11 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:05:21 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:05:21 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:05:31 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:05:31 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:05:41 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:05:41 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:05:51 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:05:51 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:06:01 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:06:01 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:06:11 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:06:11 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:06:21 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:06:21 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:06:31 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:06:31 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:06:41 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:06:41 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:06:51 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:06:51 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:07:01 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:07:01 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:07:11 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:07:11 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:07:21 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:07:21 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:07:31 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:07:31 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:07:41 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:07:41 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:07:51 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:07:51 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:08:01 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:08:01 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:08:11 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:08:11 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:08:21 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:08:21 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:08:31 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:08:31 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:08:41 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:08:41 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:08:51 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:08:51 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:09:01 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:09:01 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:09:11 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:09:11 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:09:21 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:09:21 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:09:31 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:09:31 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:09:41 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:09:41 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:09:51 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:09:51 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:10:01 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:10:01 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:10:11 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:10:11 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:10:21 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:10:21 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:10:31 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:10:31 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:10:41 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:10:41 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:10:51 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:10:51 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:11:01 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:11:01 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:11:11 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:11:11 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:11:21 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:11:21 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:11:31 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:11:31 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:11:41 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:11:41 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:11:51 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:11:51 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:12:01 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:12:01 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:12:11 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:12:11 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:12:21 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:12:21 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:12:31 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:12:31 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:12:41 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:12:41 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:12:51 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:12:51 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:13:01 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:13:01 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:13:11 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:13:11 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:13:21 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:13:21 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:13:31 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:13:31 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:13:41 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:13:41 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:13:51 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:13:51 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:14:01 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:14:01 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:14:11 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:14:11 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:14:21 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:14:21 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:14:31 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:14:31 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:14:41 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:14:41 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
|
INFO 07-14 11:14:51 metrics.py:345] Avg prompt throughput: 0.0 tokens/s, Avg generation throughput: 0.0 tokens/s, Running: 0 reqs, Swapped: 0 reqs, Pending: 0 reqs, GPU KV cache usage: 0.0%, CPU KV cache usage: 0.0%.
|
||||||
|
INFO 07-14 11:14:51 metrics.py:361] Prefix cache hit rate: GPU: 85.71%, CPU: 0.00%
|
||||||
@@ -0,0 +1,104 @@
|
|||||||
|
/usr/local/corex/lib/python3/dist-packages/torch/cuda/__init__.py:51: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you.
|
||||||
|
import pynvml # type: ignore[import]
|
||||||
|
INFO 07-14 10:45:59 importing.py:10] Triton not installed; certain GPU-related functions will not be available.
|
||||||
|
2026-07-14 10:46:01.653483: I tensorflow/core/util/port.cc:110] oneDNN custom operations are on. You may see slightly different numerical results due to floating-point round-off errors from different computation orders. To turn them off, set the environment variable `TF_ENABLE_ONEDNN_OPTS=0`.
|
||||||
|
2026-07-14 10:46:01.708498: I tensorflow/core/platform/cpu_feature_guard.cc:182] This TensorFlow binary is optimized to use available CPU instructions in performance-critical operations.
|
||||||
|
To enable the following instructions: SSE3 SSE4.1 SSE4.2 AVX AVX2 AVX512F AVX512_VNNI AVX512_BF16 AVX_VNNI AMX_TILE AMX_INT8 AMX_BF16 FMA, in other operations, rebuild TensorFlow with the appropriate compiler flags.
|
||||||
|
WARNING:tensorflow:Deprecation warnings have been disabled. Set TF_ENABLE_DEPRECATION_WARNINGS=1 to re-enable them.
|
||||||
|
INFO 07-14 10:46:07 api_server.py:530] vLLM API server version 0.6.3
|
||||||
|
INFO 07-14 10:46:07 api_server.py:531] args: Namespace(host='0.0.0.0', port=1111, uvicorn_log_level='info', allow_credentials=False, allowed_origins=['*'], allowed_methods=['*'], allowed_headers=['*'], api_key=None, lora_modules=None, prompt_adapters=None, chat_template=None, response_role='assistant', ssl_keyfile=None, ssl_certfile=None, ssl_ca_certs=None, ssl_cert_reqs=0, root_path=None, middleware=[], return_tokens_as_token_ids=False, disable_frontend_multiprocessing=True, enable_auto_tool_choice=True, tool_call_parser='qwen3_coder', tool_parser_plugin='', reasoning_parser='qwen3', model='/root/public-storage/models/Qwen/Qwen3.6-35B-A3B', tokenizer=None, skip_tokenizer_init=False, revision=None, code_revision=None, tokenizer_revision=None, tokenizer_mode='auto', trust_remote_code=True, download_dir=None, load_format='auto', config_format='auto', dtype='auto', kv_cache_dtype='auto', quantization_param_path=None, max_model_len=100000, guided_decoding_backend='outlines', distributed_executor_backend=None, worker_use_ray=False, pipeline_parallel_size=1, tensor_parallel_size=4, max_parallel_loading_workers=None, ray_workers_use_nsight=False, block_size=16, enable_prefix_caching=True, disable_sliding_window=False, use_v2_block_manager=True, num_lookahead_slots=0, seed=0, swap_space=4, cpu_offload_gb=0, gpu_memory_utilization=0.95, num_gpu_blocks_override=None, max_num_batched_tokens=8192, max_num_seqs=2, max_logprobs=20, disable_log_stats=False, quantization=None, rope_scaling=None, rope_theta=None, enforce_eager=False, max_context_len_to_capture=None, max_seq_len_to_capture=32768, disable_custom_all_reduce=False, tokenizer_pool_size=0, tokenizer_pool_type='ray', tokenizer_pool_extra_config=None, limit_mm_per_prompt=None, mm_processor_kwargs=None, enable_lora=False, max_loras=1, max_lora_rank=16, lora_extra_vocab_size=256, lora_dtype='auto', long_lora_scaling_factors=None, max_cpu_loras=None, fully_sharded_loras=False, enable_prompt_adapter=False, max_prompt_adapters=1, max_prompt_adapter_token=0, device='auto', num_scheduler_steps=1, multi_step_stream_outputs=True, scheduler_delay_factor=0.0, enable_chunked_prefill=True, speculative_model=None, speculative_model_quantization=None, num_speculative_tokens=None, speculative_disable_mqa_scorer=False, speculative_draft_tensor_parallel_size=None, speculative_max_model_len=None, speculative_disable_by_batch_size=None, ngram_prompt_lookup_max=None, ngram_prompt_lookup_min=None, spec_decoding_acceptance_method='rejection_sampler', typical_acceptance_sampler_posterior_threshold=None, typical_acceptance_sampler_posterior_alpha=None, disable_logprobs_during_spec_decoding=None, model_loader_extra_config=None, ignore_patterns=[], preemption_mode=None, served_model_name=['llm'], qlora_adapter_name_or_path=None, otlp_traces_endpoint=None, collect_detailed_traces=None, disable_async_output_proc=False, override_neuron_config=None, scheduling_policy='fcfs', disable_log_requests=True, max_log_len=None, disable_fastapi_docs=False)
|
||||||
|
INFO 07-14 10:46:07 config.py:1670] Downcasting torch.float32 to torch.float16.
|
||||||
|
INFO 07-14 10:46:18 config.py:887] Defaulting to use mp for distributed inference
|
||||||
|
INFO 07-14 10:46:18 config.py:1005] Chunked prefill is enabled with max_num_batched_tokens=8192.
|
||||||
|
INFO 07-14 10:46:18 llm_engine.py:237] Initializing an LLM engine (v0.6.3) with config: model='/root/public-storage/models/Qwen/Qwen3.6-35B-A3B', speculative_config=None, tokenizer='/root/public-storage/models/Qwen/Qwen3.6-35B-A3B', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, override_neuron_config=None, rope_scaling=None, rope_theta=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.float16, max_seq_len=100000, download_dir=None, load_format=LoadFormat.AUTO, tensor_parallel_size=4, pipeline_parallel_size=1, disable_custom_all_reduce=False, quantization=None, enforce_eager=False, kv_cache_dtype=auto, quantization_param_path=None, device_config=cuda, decoding_config=DecodingConfig(guided_decoding_backend='outlines'), observability_config=ObservabilityConfig(otlp_traces_endpoint=None, collect_model_forward_time=False, collect_model_execute_time=False), seed=0, served_model_name=llm, use_v2_block_manager=True, num_scheduler_steps=1, chunked_prefill_enabled=True multi_step_stream_outputs=True, enable_prefix_caching=True, use_async_output_proc=True, use_cached_outputs=False, mm_processor_kwargs=None)
|
||||||
|
WARNING 07-14 10:46:18 multiproc_gpu_executor.py:53] Reducing Torch parallelism from 64 threads to 1 to avoid unnecessary CPU contention. Set OMP_NUM_THREADS in the external environment to tune this value as needed.
|
||||||
|
INFO 07-14 10:46:18 custom_cache_manager.py:17] Setting Triton cache manager to: vllm.triton_utils.custom_cache_manager:CustomCacheManager
|
||||||
|
INFO 07-14 10:46:18 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
INFO 07-14 10:46:18 selector.py:115] Using XFormers backend.
|
||||||
|
/usr/local/corex/lib/python3/dist-packages/torch/cuda/__init__.py:51: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you.
|
||||||
|
import pynvml # type: ignore[import]
|
||||||
|
/usr/local/corex/lib/python3/dist-packages/torch/cuda/__init__.py:51: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you.
|
||||||
|
import pynvml # type: ignore[import]
|
||||||
|
/usr/local/corex/lib/python3/dist-packages/torch/cuda/__init__.py:51: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you.
|
||||||
|
import pynvml # type: ignore[import]
|
||||||
|
INFO 07-14 10:46:20 importing.py:10] Triton not installed; certain GPU-related functions will not be available.
|
||||||
|
INFO 07-14 10:46:20 importing.py:10] Triton not installed; certain GPU-related functions will not be available.
|
||||||
|
INFO 07-14 10:46:20 importing.py:10] Triton not installed; certain GPU-related functions will not be available.
|
||||||
|
WARNING:tensorflow:Deprecation warnings have been disabled. Set TF_ENABLE_DEPRECATION_WARNINGS=1 to re-enable them.
|
||||||
|
WARNING:tensorflow:Deprecation warnings have been disabled. Set TF_ENABLE_DEPRECATION_WARNINGS=1 to re-enable them.
|
||||||
|
WARNING:tensorflow:Deprecation warnings have been disabled. Set TF_ENABLE_DEPRECATION_WARNINGS=1 to re-enable them.
|
||||||
|
[1;36m(VllmWorkerProcess pid=8115)[0;0m INFO 07-14 10:46:28 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=8115)[0;0m INFO 07-14 10:46:28 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=8114)[0;0m INFO 07-14 10:46:28 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=8114)[0;0m INFO 07-14 10:46:28 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=8115)[0;0m INFO 07-14 10:46:28 multiproc_worker_utils.py:216] Worker ready; awaiting tasks
|
||||||
|
[1;36m(VllmWorkerProcess pid=8114)[0;0m INFO 07-14 10:46:28 multiproc_worker_utils.py:216] Worker ready; awaiting tasks
|
||||||
|
[1;36m(VllmWorkerProcess pid=8116)[0;0m INFO 07-14 10:46:28 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=8116)[0;0m INFO 07-14 10:46:28 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=8116)[0;0m INFO 07-14 10:46:28 multiproc_worker_utils.py:216] Worker ready; awaiting tasks
|
||||||
|
INFO 07-14 10:46:28 shm_broadcast.py:242] vLLM message queue communication handle: Handle(connect_ip='127.0.0.1', local_reader_ranks=[1, 2, 3], buffer=<vllm.distributed.device_communicators.shm_broadcast.ShmRingBuffer object at 0x7fdf16a327a0>, local_subscribe_port=53133, remote_subscribe_port=None)
|
||||||
|
INFO 07-14 10:46:28 model_runner.py:1065] Starting to load model /root/public-storage/models/Qwen/Qwen3.6-35B-A3B...
|
||||||
|
[1;36m(VllmWorkerProcess pid=8115)[0;0m INFO 07-14 10:46:28 model_runner.py:1065] Starting to load model /root/public-storage/models/Qwen/Qwen3.6-35B-A3B...
|
||||||
|
[1;36m(VllmWorkerProcess pid=8114)[0;0m INFO 07-14 10:46:28 model_runner.py:1065] Starting to load model /root/public-storage/models/Qwen/Qwen3.6-35B-A3B...
|
||||||
|
[1;36m(VllmWorkerProcess pid=8116)[0;0m INFO 07-14 10:46:28 model_runner.py:1065] Starting to load model /root/public-storage/models/Qwen/Qwen3.6-35B-A3B...
|
||||||
|
INFO 07-14 10:46:28 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
INFO 07-14 10:46:28 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=8116)[0;0m INFO 07-14 10:46:28 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=8114)[0;0m INFO 07-14 10:46:28 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=8115)[0;0m INFO 07-14 10:46:28 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
[1;36m(VllmWorkerProcess pid=8116)[0;0m INFO 07-14 10:46:28 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=8114)[0;0m INFO 07-14 10:46:28 selector.py:115] Using XFormers backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=8115)[0;0m INFO 07-14 10:46:28 selector.py:115] Using XFormers backend.
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 0% Completed | 0/26 [00:00<?, ?it/s]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 4% Completed | 1/26 [00:01<00:41, 1.65s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 8% Completed | 2/26 [00:02<00:21, 1.12it/s]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 12% Completed | 3/26 [00:03<00:30, 1.31s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 15% Completed | 4/26 [00:05<00:34, 1.58s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 19% Completed | 5/26 [00:06<00:29, 1.39s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 23% Completed | 6/26 [00:08<00:32, 1.63s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 27% Completed | 7/26 [00:09<00:23, 1.25s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 31% Completed | 8/26 [00:11<00:25, 1.43s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 35% Completed | 9/26 [00:12<00:22, 1.30s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 38% Completed | 10/26 [00:14<00:24, 1.54s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 42% Completed | 11/26 [00:14<00:17, 1.16s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 46% Completed | 12/26 [00:16<00:19, 1.38s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 50% Completed | 13/26 [00:18<00:20, 1.55s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 54% Completed | 14/26 [00:20<00:20, 1.74s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 58% Completed | 15/26 [00:21<00:16, 1.47s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 62% Completed | 16/26 [00:23<00:17, 1.75s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 65% Completed | 17/26 [00:24<00:12, 1.38s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 69% Completed | 18/26 [00:26<00:12, 1.52s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 73% Completed | 19/26 [00:27<00:11, 1.58s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 77% Completed | 20/26 [00:30<00:11, 1.86s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 81% Completed | 21/26 [00:31<00:08, 1.63s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 85% Completed | 22/26 [00:32<00:05, 1.38s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 88% Completed | 23/26 [00:34<00:05, 1.69s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 92% Completed | 24/26 [00:36<00:03, 1.78s/it]
|
||||||
|
[1;36m(VllmWorkerProcess pid=8116)[0;0m INFO 07-14 10:47:05 model_runner.py:1076] Loading model weights took 16.2303 GB
|
||||||
|
[1;36m(VllmWorkerProcess pid=8114)[0;0m INFO 07-14 10:47:05 model_runner.py:1076] Loading model weights took 16.2303 GB
|
||||||
|
[1;36m(VllmWorkerProcess pid=8115)[0;0m INFO 07-14 10:47:05 model_runner.py:1076] Loading model weights took 16.2303 GB
|
||||||
@@ -0,0 +1,41 @@
|
|||||||
|
/usr/local/corex/lib/python3/dist-packages/torch/cuda/__init__.py:51: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you.
|
||||||
|
import pynvml # type: ignore[import]
|
||||||
|
INFO 07-14 09:19:28 importing.py:10] Triton not installed; certain GPU-related functions will not be available.
|
||||||
|
2026-07-14 09:19:30.064377: I tensorflow/core/util/port.cc:110] oneDNN custom operations are on. You may see slightly different numerical results due to floating-point round-off errors from different computation orders. To turn them off, set the environment variable `TF_ENABLE_ONEDNN_OPTS=0`.
|
||||||
|
2026-07-14 09:19:30.115922: I tensorflow/core/platform/cpu_feature_guard.cc:182] This TensorFlow binary is optimized to use available CPU instructions in performance-critical operations.
|
||||||
|
To enable the following instructions: SSE3 SSE4.1 SSE4.2 AVX AVX2 AVX512F AVX512_VNNI AVX512_BF16 AVX_VNNI AMX_TILE AMX_INT8 AMX_BF16 FMA, in other operations, rebuild TensorFlow with the appropriate compiler flags.
|
||||||
|
WARNING:tensorflow:Deprecation warnings have been disabled. Set TF_ENABLE_DEPRECATION_WARNINGS=1 to re-enable them.
|
||||||
|
INFO 07-14 09:19:35 api_server.py:530] vLLM API server version 0.6.3
|
||||||
|
INFO 07-14 09:19:35 api_server.py:531] args: Namespace(host='0.0.0.0', port=1111, uvicorn_log_level='info', allow_credentials=False, allowed_origins=['*'], allowed_methods=['*'], allowed_headers=['*'], api_key=None, lora_modules=None, prompt_adapters=None, chat_template=None, response_role='assistant', ssl_keyfile=None, ssl_certfile=None, ssl_ca_certs=None, ssl_cert_reqs=0, root_path=None, middleware=[], return_tokens_as_token_ids=False, disable_frontend_multiprocessing=True, enable_auto_tool_choice=True, tool_call_parser='qwen3_coder', tool_parser_plugin='', reasoning_parser='qwen3', model='/root/public-storage/models/Qwen/Qwen3.6-35B-A3B', tokenizer=None, skip_tokenizer_init=False, revision=None, code_revision=None, tokenizer_revision=None, tokenizer_mode='auto', trust_remote_code=True, download_dir=None, load_format='auto', config_format='auto', dtype='auto', kv_cache_dtype='auto', quantization_param_path=None, max_model_len=100000, guided_decoding_backend='outlines', distributed_executor_backend=None, worker_use_ray=False, pipeline_parallel_size=1, tensor_parallel_size=4, max_parallel_loading_workers=None, ray_workers_use_nsight=False, block_size=16, enable_prefix_caching=True, disable_sliding_window=False, use_v2_block_manager=True, num_lookahead_slots=0, seed=0, swap_space=4, cpu_offload_gb=0, gpu_memory_utilization=0.95, num_gpu_blocks_override=None, max_num_batched_tokens=8192, max_num_seqs=2, max_logprobs=20, disable_log_stats=False, quantization=None, rope_scaling=None, rope_theta=None, enforce_eager=False, max_context_len_to_capture=None, max_seq_len_to_capture=32768, disable_custom_all_reduce=False, tokenizer_pool_size=0, tokenizer_pool_type='ray', tokenizer_pool_extra_config=None, limit_mm_per_prompt=None, mm_processor_kwargs=None, enable_lora=False, max_loras=1, max_lora_rank=16, lora_extra_vocab_size=256, lora_dtype='auto', long_lora_scaling_factors=None, max_cpu_loras=None, fully_sharded_loras=False, enable_prompt_adapter=False, max_prompt_adapters=1, max_prompt_adapter_token=0, device='auto', num_scheduler_steps=4, multi_step_stream_outputs=True, scheduler_delay_factor=0.0, enable_chunked_prefill=None, speculative_model=None, speculative_model_quantization=None, num_speculative_tokens=None, speculative_disable_mqa_scorer=False, speculative_draft_tensor_parallel_size=None, speculative_max_model_len=None, speculative_disable_by_batch_size=None, ngram_prompt_lookup_max=None, ngram_prompt_lookup_min=None, spec_decoding_acceptance_method='rejection_sampler', typical_acceptance_sampler_posterior_threshold=None, typical_acceptance_sampler_posterior_alpha=None, disable_logprobs_during_spec_decoding=None, model_loader_extra_config=None, ignore_patterns=[], preemption_mode=None, served_model_name=['llm'], qlora_adapter_name_or_path=None, otlp_traces_endpoint=None, collect_detailed_traces=None, disable_async_output_proc=False, override_neuron_config=None, scheduling_policy='fcfs', disable_log_requests=True, max_log_len=None, disable_fastapi_docs=False)
|
||||||
|
INFO 07-14 09:19:35 config.py:1670] Downcasting torch.float32 to torch.float16.
|
||||||
|
INFO 07-14 09:19:46 config.py:887] Defaulting to use mp for distributed inference
|
||||||
|
WARNING 07-14 09:19:46 arg_utils.py:963] The model has a long context length (100000). This may cause OOM errors during the initial memory profiling phase, or result in low performance due to small KV cache space. Consider setting --max-model-len to a smaller value.
|
||||||
|
Traceback (most recent call last):
|
||||||
|
File "/usr/local/lib/python3.10/runpy.py", line 196, in _run_module_as_main
|
||||||
|
return _run_code(code, main_globals, None,
|
||||||
|
File "/usr/local/lib/python3.10/runpy.py", line 86, in _run_code
|
||||||
|
exec(code, run_globals)
|
||||||
|
File "/usr/local/corex/lib/python3/dist-packages/vllm/entrypoints/openai/api_server.py", line 595, in <module>
|
||||||
|
uvloop.run(run_server(args))
|
||||||
|
File "/usr/local/lib/python3.10/site-packages/uvloop/__init__.py", line 82, in run
|
||||||
|
return loop.run_until_complete(wrapper())
|
||||||
|
File "uvloop/loop.pyx", line 1518, in uvloop.loop.Loop.run_until_complete
|
||||||
|
File "/usr/local/lib/python3.10/site-packages/uvloop/__init__.py", line 61, in wrapper
|
||||||
|
return await main
|
||||||
|
File "/usr/local/corex/lib/python3/dist-packages/vllm/entrypoints/openai/api_server.py", line 562, in run_server
|
||||||
|
async with build_async_engine_client(args) as engine_client:
|
||||||
|
File "/usr/local/lib/python3.10/contextlib.py", line 199, in __aenter__
|
||||||
|
return await anext(self.gen)
|
||||||
|
File "/usr/local/corex/lib/python3/dist-packages/vllm/entrypoints/openai/api_server.py", line 108, in build_async_engine_client
|
||||||
|
async with build_async_engine_client_from_engine_args(
|
||||||
|
File "/usr/local/lib/python3.10/contextlib.py", line 199, in __aenter__
|
||||||
|
return await anext(self.gen)
|
||||||
|
File "/usr/local/corex/lib/python3/dist-packages/vllm/entrypoints/openai/api_server.py", line 130, in build_async_engine_client_from_engine_args
|
||||||
|
engine_config = engine_args.create_engine_config()
|
||||||
|
File "/usr/local/corex/lib/python3/dist-packages/vllm/engine/arg_utils.py", line 1021, in create_engine_config
|
||||||
|
scheduler_config = SchedulerConfig(
|
||||||
|
File "/usr/local/corex/lib/python3/dist-packages/vllm/config.py", line 1021, in __init__
|
||||||
|
self._verify_args()
|
||||||
|
File "/usr/local/corex/lib/python3/dist-packages/vllm/config.py", line 1026, in _verify_args
|
||||||
|
raise ValueError(
|
||||||
|
ValueError: max_num_batched_tokens (8192) is smaller than max_model_len (100000). This effectively limits the maximum sequence length to max_num_batched_tokens and makes vLLM reject longer sequences. Please increase max_num_batched_tokens or decrease max_model_len.
|
||||||
@@ -0,0 +1,212 @@
|
|||||||
|
/usr/local/corex/lib/python3/dist-packages/torch/cuda/__init__.py:51: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you.
|
||||||
|
import pynvml # type: ignore[import]
|
||||||
|
INFO 07-14 09:43:16 importing.py:10] Triton not installed; certain GPU-related functions will not be available.
|
||||||
|
2026-07-14 09:43:17.917732: I tensorflow/core/util/port.cc:110] oneDNN custom operations are on. You may see slightly different numerical results due to floating-point round-off errors from different computation orders. To turn them off, set the environment variable `TF_ENABLE_ONEDNN_OPTS=0`.
|
||||||
|
2026-07-14 09:43:17.968433: I tensorflow/core/platform/cpu_feature_guard.cc:182] This TensorFlow binary is optimized to use available CPU instructions in performance-critical operations.
|
||||||
|
To enable the following instructions: SSE3 SSE4.1 SSE4.2 AVX AVX2 AVX512F AVX512_VNNI AVX512_BF16 AVX_VNNI AMX_TILE AMX_INT8 AMX_BF16 FMA, in other operations, rebuild TensorFlow with the appropriate compiler flags.
|
||||||
|
WARNING:tensorflow:Deprecation warnings have been disabled. Set TF_ENABLE_DEPRECATION_WARNINGS=1 to re-enable them.
|
||||||
|
INFO 07-14 09:43:23 api_server.py:530] vLLM API server version 0.6.3
|
||||||
|
INFO 07-14 09:43:23 api_server.py:531] args: Namespace(host='0.0.0.0', port=1111, uvicorn_log_level='info', allow_credentials=False, allowed_origins=['*'], allowed_methods=['*'], allowed_headers=['*'], api_key=None, lora_modules=None, prompt_adapters=None, chat_template=None, response_role='assistant', ssl_keyfile=None, ssl_certfile=None, ssl_ca_certs=None, ssl_cert_reqs=0, root_path=None, middleware=[], return_tokens_as_token_ids=False, disable_frontend_multiprocessing=True, enable_auto_tool_choice=True, tool_call_parser='qwen3_coder', tool_parser_plugin='', reasoning_parser='qwen3', model='/root/public-storage/models/Qwen/Qwen3.6-35B-A3B', tokenizer=None, skip_tokenizer_init=False, revision=None, code_revision=None, tokenizer_revision=None, tokenizer_mode='auto', trust_remote_code=True, download_dir=None, load_format='auto', config_format='auto', dtype='auto', kv_cache_dtype='auto', quantization_param_path=None, max_model_len=4096, guided_decoding_backend='outlines', distributed_executor_backend=None, worker_use_ray=False, pipeline_parallel_size=1, tensor_parallel_size=4, max_parallel_loading_workers=None, ray_workers_use_nsight=False, block_size=16, enable_prefix_caching=True, disable_sliding_window=False, use_v2_block_manager=True, num_lookahead_slots=0, seed=0, swap_space=4, cpu_offload_gb=0, gpu_memory_utilization=0.95, num_gpu_blocks_override=None, max_num_batched_tokens=8192, max_num_seqs=2, max_logprobs=20, disable_log_stats=False, quantization=None, rope_scaling=None, rope_theta=None, enforce_eager=False, max_context_len_to_capture=None, max_seq_len_to_capture=4096, disable_custom_all_reduce=False, tokenizer_pool_size=0, tokenizer_pool_type='ray', tokenizer_pool_extra_config=None, limit_mm_per_prompt=None, mm_processor_kwargs=None, enable_lora=False, max_loras=1, max_lora_rank=16, lora_extra_vocab_size=256, lora_dtype='auto', long_lora_scaling_factors=None, max_cpu_loras=None, fully_sharded_loras=False, enable_prompt_adapter=False, max_prompt_adapters=1, max_prompt_adapter_token=0, device='auto', num_scheduler_steps=4, multi_step_stream_outputs=True, scheduler_delay_factor=0.0, enable_chunked_prefill=None, speculative_model=None, speculative_model_quantization=None, num_speculative_tokens=None, speculative_disable_mqa_scorer=False, speculative_draft_tensor_parallel_size=None, speculative_max_model_len=None, speculative_disable_by_batch_size=None, ngram_prompt_lookup_max=None, ngram_prompt_lookup_min=None, spec_decoding_acceptance_method='rejection_sampler', typical_acceptance_sampler_posterior_threshold=None, typical_acceptance_sampler_posterior_alpha=None, disable_logprobs_during_spec_decoding=None, model_loader_extra_config=None, ignore_patterns=[], preemption_mode=None, served_model_name=['llm'], qlora_adapter_name_or_path=None, otlp_traces_endpoint=None, collect_detailed_traces=None, disable_async_output_proc=False, override_neuron_config=None, scheduling_policy='fcfs', disable_log_requests=True, max_log_len=None, disable_fastapi_docs=False)
|
||||||
|
INFO 07-14 09:43:23 config.py:1670] Downcasting torch.float32 to torch.float16.
|
||||||
|
INFO 07-14 09:43:34 config.py:887] Defaulting to use mp for distributed inference
|
||||||
|
WARNING 07-14 09:43:34 config.py:380] To see benefits of async output processing, enable CUDA graph. Since, enforce-eager is enabled, async output processor cannot be used
|
||||||
|
INFO 07-14 09:43:34 llm_engine.py:237] Initializing an LLM engine (v0.6.3) with config: model='/root/public-storage/models/Qwen/Qwen3.6-35B-A3B', speculative_config=None, tokenizer='/root/public-storage/models/Qwen/Qwen3.6-35B-A3B', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, override_neuron_config=None, rope_scaling=None, rope_theta=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.float16, max_seq_len=4096, download_dir=None, load_format=LoadFormat.AUTO, tensor_parallel_size=4, pipeline_parallel_size=1, disable_custom_all_reduce=True, quantization=None, enforce_eager=True, kv_cache_dtype=auto, quantization_param_path=None, device_config=cuda, decoding_config=DecodingConfig(guided_decoding_backend='outlines'), observability_config=ObservabilityConfig(otlp_traces_endpoint=None, collect_model_forward_time=False, collect_model_execute_time=False), seed=0, served_model_name=llm, use_v2_block_manager=True, num_scheduler_steps=4, chunked_prefill_enabled=False multi_step_stream_outputs=True, enable_prefix_caching=True, use_async_output_proc=False, use_cached_outputs=False, mm_processor_kwargs=None)
|
||||||
|
INFO 07-14 09:43:34 custom_cache_manager.py:17] Setting Triton cache manager to: vllm.triton_utils.custom_cache_manager:CustomCacheManager
|
||||||
|
INFO 07-14 09:43:34 selector.py:141] Using Flashinfer backend.
|
||||||
|
WARNING 07-14 09:43:34 registry.py:205] `mm_limits` has already been set for model=/root/public-storage/models/Qwen/Qwen3.6-35B-A3B, and will be overwritten by the new values.
|
||||||
|
/usr/local/corex/lib/python3/dist-packages/torch/cuda/__init__.py:51: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you.
|
||||||
|
import pynvml # type: ignore[import]
|
||||||
|
/usr/local/corex/lib/python3/dist-packages/torch/cuda/__init__.py:51: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you.
|
||||||
|
import pynvml # type: ignore[import]
|
||||||
|
/usr/local/corex/lib/python3/dist-packages/torch/cuda/__init__.py:51: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you.
|
||||||
|
import pynvml # type: ignore[import]
|
||||||
|
INFO 07-14 09:43:36 importing.py:10] Triton not installed; certain GPU-related functions will not be available.
|
||||||
|
INFO 07-14 09:43:36 importing.py:10] Triton not installed; certain GPU-related functions will not be available.
|
||||||
|
INFO 07-14 09:43:36 importing.py:10] Triton not installed; certain GPU-related functions will not be available.
|
||||||
|
WARNING:tensorflow:Deprecation warnings have been disabled. Set TF_ENABLE_DEPRECATION_WARNINGS=1 to re-enable them.
|
||||||
|
WARNING:tensorflow:Deprecation warnings have been disabled. Set TF_ENABLE_DEPRECATION_WARNINGS=1 to re-enable them.
|
||||||
|
WARNING:tensorflow:Deprecation warnings have been disabled. Set TF_ENABLE_DEPRECATION_WARNINGS=1 to re-enable them.
|
||||||
|
[1;36m(VllmWorkerProcess pid=6761)[0;0m INFO 07-14 09:43:44 selector.py:141] Using Flashinfer backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=6761)[0;0m WARNING 07-14 09:43:44 registry.py:205] `mm_limits` has already been set for model=/root/public-storage/models/Qwen/Qwen3.6-35B-A3B, and will be overwritten by the new values.
|
||||||
|
[1;36m(VllmWorkerProcess pid=6761)[0;0m INFO 07-14 09:43:44 multiproc_worker_utils.py:216] Worker ready; awaiting tasks
|
||||||
|
[1;36m(VllmWorkerProcess pid=6760)[0;0m INFO 07-14 09:43:44 selector.py:141] Using Flashinfer backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=6760)[0;0m WARNING 07-14 09:43:44 registry.py:205] `mm_limits` has already been set for model=/root/public-storage/models/Qwen/Qwen3.6-35B-A3B, and will be overwritten by the new values.
|
||||||
|
[1;36m(VllmWorkerProcess pid=6760)[0;0m INFO 07-14 09:43:44 multiproc_worker_utils.py:216] Worker ready; awaiting tasks
|
||||||
|
[1;36m(VllmWorkerProcess pid=6759)[0;0m INFO 07-14 09:43:44 selector.py:141] Using Flashinfer backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=6759)[0;0m WARNING 07-14 09:43:44 registry.py:205] `mm_limits` has already been set for model=/root/public-storage/models/Qwen/Qwen3.6-35B-A3B, and will be overwritten by the new values.
|
||||||
|
[1;36m(VllmWorkerProcess pid=6759)[0;0m INFO 07-14 09:43:44 multiproc_worker_utils.py:216] Worker ready; awaiting tasks
|
||||||
|
INFO 07-14 09:43:45 shm_broadcast.py:242] vLLM message queue communication handle: Handle(connect_ip='127.0.0.1', local_reader_ranks=[1, 2, 3], buffer=<vllm.distributed.device_communicators.shm_broadcast.ShmRingBuffer object at 0x7f56a01879d0>, local_subscribe_port=60197, remote_subscribe_port=None)
|
||||||
|
INFO 07-14 09:43:45 model_runner.py:1065] Starting to load model /root/public-storage/models/Qwen/Qwen3.6-35B-A3B...
|
||||||
|
[1;36m(VllmWorkerProcess pid=6759)[0;0m INFO 07-14 09:43:45 model_runner.py:1065] Starting to load model /root/public-storage/models/Qwen/Qwen3.6-35B-A3B...
|
||||||
|
[1;36m(VllmWorkerProcess pid=6760)[0;0m INFO 07-14 09:43:45 model_runner.py:1065] Starting to load model /root/public-storage/models/Qwen/Qwen3.6-35B-A3B...
|
||||||
|
[1;36m(VllmWorkerProcess pid=6761)[0;0m INFO 07-14 09:43:45 model_runner.py:1065] Starting to load model /root/public-storage/models/Qwen/Qwen3.6-35B-A3B...
|
||||||
|
[1;36m(VllmWorkerProcess pid=6759)[0;0m INFO 07-14 09:43:45 selector.py:141] Using Flashinfer backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=6761)[0;0m INFO 07-14 09:43:45 selector.py:141] Using Flashinfer backend.
|
||||||
|
[1;36m(VllmWorkerProcess pid=6760)[0;0m INFO 07-14 09:43:45 selector.py:141] Using Flashinfer backend.
|
||||||
|
INFO 07-14 09:43:45 selector.py:141] Using Flashinfer backend.
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 0% Completed | 0/26 [00:00<?, ?it/s]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 4% Completed | 1/26 [00:01<00:45, 1.81s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 8% Completed | 2/26 [00:02<00:24, 1.02s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 12% Completed | 3/26 [00:04<00:32, 1.41s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 15% Completed | 4/26 [00:06<00:34, 1.59s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 19% Completed | 5/26 [00:06<00:28, 1.36s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 23% Completed | 6/26 [00:09<00:32, 1.60s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 27% Completed | 7/26 [00:09<00:23, 1.22s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 31% Completed | 8/26 [00:11<00:25, 1.42s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 35% Completed | 9/26 [00:12<00:21, 1.29s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 38% Completed | 10/26 [00:14<00:24, 1.52s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 42% Completed | 11/26 [00:14<00:16, 1.13s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 46% Completed | 12/26 [00:16<00:18, 1.35s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 50% Completed | 13/26 [00:18<00:19, 1.52s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 54% Completed | 14/26 [00:20<00:20, 1.71s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 58% Completed | 15/26 [00:21<00:15, 1.44s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 62% Completed | 16/26 [00:23<00:17, 1.72s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 65% Completed | 17/26 [00:24<00:12, 1.35s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 69% Completed | 18/26 [00:26<00:11, 1.50s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 73% Completed | 19/26 [00:27<00:10, 1.56s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 77% Completed | 20/26 [00:30<00:10, 1.82s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 81% Completed | 21/26 [00:31<00:07, 1.59s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 85% Completed | 22/26 [00:31<00:05, 1.34s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 88% Completed | 23/26 [00:34<00:04, 1.65s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 92% Completed | 24/26 [00:36<00:03, 1.75s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 96% Completed | 25/26 [00:38<00:01, 1.82s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 100% Completed | 26/26 [00:38<00:00, 1.44s/it]
|
||||||
|
|
||||||
|
Loading safetensors checkpoint shards: 100% Completed | 26/26 [00:38<00:00, 1.49s/it]
|
||||||
|
|
||||||
|
INFO 07-14 09:44:24 model_runner.py:1076] Loading model weights took 16.2303 GB
|
||||||
|
[1;36m(VllmWorkerProcess pid=6759)[0;0m INFO 07-14 09:44:24 model_runner.py:1076] Loading model weights took 16.2303 GB
|
||||||
|
[1;36m(VllmWorkerProcess pid=6760)[0;0m INFO 07-14 09:44:24 model_runner.py:1076] Loading model weights took 16.2303 GB
|
||||||
|
[1;36m(VllmWorkerProcess pid=6761)[0;0m INFO 07-14 09:44:24 model_runner.py:1076] Loading model weights took 16.2303 GB
|
||||||
|
[1;36m(VllmWorkerProcess pid=6761)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] Exception in worker VllmWorkerProcess while processing method determine_num_available_blocks: 'NoneType' object is not callable, Traceback (most recent call last):
|
||||||
|
[1;36m(VllmWorkerProcess pid=6761)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] File "/usr/local/corex/lib/python3/dist-packages/vllm/executor/multiproc_worker_utils.py", line 224, in _run_worker_process
|
||||||
|
[1;36m(VllmWorkerProcess pid=6761)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] output = executor(*args, **kwargs)
|
||||||
|
[1;36m(VllmWorkerProcess pid=6761)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] File "/usr/local/corex/lib/python3/dist-packages/torch/utils/_contextlib.py", line 115, in decorate_context
|
||||||
|
[1;36m(VllmWorkerProcess pid=6761)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] return func(*args, **kwargs)
|
||||||
|
[1;36m(VllmWorkerProcess pid=6761)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] File "/usr/local/corex/lib/python3/dist-packages/vllm/worker/worker.py", line 223, in determine_num_available_blocks
|
||||||
|
[1;36m(VllmWorkerProcess pid=6761)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] self.model_runner.profile_run()
|
||||||
|
[1;36m(VllmWorkerProcess pid=6761)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] File "/usr/local/corex/lib/python3/dist-packages/vllm/worker/multi_step_model_runner.py", line 661, in profile_run
|
||||||
|
[1;36m(VllmWorkerProcess pid=6761)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] return self._base_model_runner.profile_run()
|
||||||
|
[1;36m(VllmWorkerProcess pid=6761)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] File "/usr/local/corex/lib/python3/dist-packages/torch/utils/_contextlib.py", line 115, in decorate_context
|
||||||
|
[1;36m(VllmWorkerProcess pid=6761)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] return func(*args, **kwargs)
|
||||||
|
[1;36m(VllmWorkerProcess pid=6761)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] File "/usr/local/corex/lib/python3/dist-packages/vllm/worker/model_runner.py", line 1314, in profile_run
|
||||||
|
[1;36m(VllmWorkerProcess pid=6761)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] self.execute_model(model_input, kv_caches, intermediate_tensors)
|
||||||
|
[1;36m(VllmWorkerProcess pid=6761)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] File "/usr/local/corex/lib/python3/dist-packages/torch/utils/_contextlib.py", line 115, in decorate_context
|
||||||
|
[1;36m(VllmWorkerProcess pid=6761)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] return func(*args, **kwargs)
|
||||||
|
[1;36m(VllmWorkerProcess pid=6761)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] File "/usr/local/corex/lib/python3/dist-packages/vllm/worker/model_runner.py", line 1641, in execute_model
|
||||||
|
[1;36m(VllmWorkerProcess pid=6761)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] self.attn_state.begin_forward(model_input)
|
||||||
|
[1;36m(VllmWorkerProcess pid=6761)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] File "/usr/local/corex/lib/python3/dist-packages/vllm/attention/backends/flashinfer.py", line 261, in begin_forward
|
||||||
|
[1;36m(VllmWorkerProcess pid=6761)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] model_input.attn_metadata.decode_wrapper = state._get_decode_wrapper()
|
||||||
|
[1;36m(VllmWorkerProcess pid=6761)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] File "/usr/local/corex/lib/python3/dist-packages/vllm/attention/backends/flashinfer.py", line 130, in _get_decode_wrapper
|
||||||
|
[1;36m(VllmWorkerProcess pid=6761)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] self._decode_wrapper = BatchDecodeWithPagedKVCacheWrapper(
|
||||||
|
[1;36m(VllmWorkerProcess pid=6761)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] TypeError: 'NoneType' object is not callable
|
||||||
|
[1;36m(VllmWorkerProcess pid=6761)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231]
|
||||||
|
[1;36m(VllmWorkerProcess pid=6759)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] Exception in worker VllmWorkerProcess while processing method determine_num_available_blocks: 'NoneType' object is not callable, Traceback (most recent call last):
|
||||||
|
[1;36m(VllmWorkerProcess pid=6759)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] File "/usr/local/corex/lib/python3/dist-packages/vllm/executor/multiproc_worker_utils.py", line 224, in _run_worker_process
|
||||||
|
[1;36m(VllmWorkerProcess pid=6759)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] output = executor(*args, **kwargs)
|
||||||
|
[1;36m(VllmWorkerProcess pid=6759)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] File "/usr/local/corex/lib/python3/dist-packages/torch/utils/_contextlib.py", line 115, in decorate_context
|
||||||
|
[1;36m(VllmWorkerProcess pid=6759)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] return func(*args, **kwargs)
|
||||||
|
[1;36m(VllmWorkerProcess pid=6759)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] File "/usr/local/corex/lib/python3/dist-packages/vllm/worker/worker.py", line 223, in determine_num_available_blocks
|
||||||
|
[1;36m(VllmWorkerProcess pid=6759)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] self.model_runner.profile_run()
|
||||||
|
[1;36m(VllmWorkerProcess pid=6759)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] File "/usr/local/corex/lib/python3/dist-packages/vllm/worker/multi_step_model_runner.py", line 661, in profile_run
|
||||||
|
[1;36m(VllmWorkerProcess pid=6759)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] return self._base_model_runner.profile_run()
|
||||||
|
[1;36m(VllmWorkerProcess pid=6759)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] File "/usr/local/corex/lib/python3/dist-packages/torch/utils/_contextlib.py", line 115, in decorate_context
|
||||||
|
[1;36m(VllmWorkerProcess pid=6759)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] return func(*args, **kwargs)
|
||||||
|
[1;36m(VllmWorkerProcess pid=6759)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] File "/usr/local/corex/lib/python3/dist-packages/vllm/worker/model_runner.py", line 1314, in profile_run
|
||||||
|
[1;36m(VllmWorkerProcess pid=6759)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] self.execute_model(model_input, kv_caches, intermediate_tensors)
|
||||||
|
[1;36m(VllmWorkerProcess pid=6759)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] File "/usr/local/corex/lib/python3/dist-packages/torch/utils/_contextlib.py", line 115, in decorate_context
|
||||||
|
[1;36m(VllmWorkerProcess pid=6759)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] return func(*args, **kwargs)
|
||||||
|
[1;36m(VllmWorkerProcess pid=6759)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] File "/usr/local/corex/lib/python3/dist-packages/vllm/worker/model_runner.py", line 1641, in execute_model
|
||||||
|
[1;36m(VllmWorkerProcess pid=6759)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] self.attn_state.begin_forward(model_input)
|
||||||
|
[1;36m(VllmWorkerProcess pid=6759)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] File "/usr/local/corex/lib/python3/dist-packages/vllm/attention/backends/flashinfer.py", line 261, in begin_forward
|
||||||
|
[1;36m(VllmWorkerProcess pid=6759)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] model_input.attn_metadata.decode_wrapper = state._get_decode_wrapper()
|
||||||
|
[1;36m(VllmWorkerProcess pid=6759)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] File "/usr/local/corex/lib/python3/dist-packages/vllm/attention/backends/flashinfer.py", line 130, in _get_decode_wrapper
|
||||||
|
[1;36m(VllmWorkerProcess pid=6759)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] self._decode_wrapper = BatchDecodeWithPagedKVCacheWrapper(
|
||||||
|
[1;36m(VllmWorkerProcess pid=6759)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] TypeError: 'NoneType' object is not callable
|
||||||
|
[1;36m(VllmWorkerProcess pid=6759)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231]
|
||||||
|
[1;36m(VllmWorkerProcess pid=6760)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] Exception in worker VllmWorkerProcess while processing method determine_num_available_blocks: 'NoneType' object is not callable, Traceback (most recent call last):
|
||||||
|
[1;36m(VllmWorkerProcess pid=6760)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] File "/usr/local/corex/lib/python3/dist-packages/vllm/executor/multiproc_worker_utils.py", line 224, in _run_worker_process
|
||||||
|
[1;36m(VllmWorkerProcess pid=6760)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] output = executor(*args, **kwargs)
|
||||||
|
[1;36m(VllmWorkerProcess pid=6760)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] File "/usr/local/corex/lib/python3/dist-packages/torch/utils/_contextlib.py", line 115, in decorate_context
|
||||||
|
[1;36m(VllmWorkerProcess pid=6760)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] return func(*args, **kwargs)
|
||||||
|
[1;36m(VllmWorkerProcess pid=6760)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] File "/usr/local/corex/lib/python3/dist-packages/vllm/worker/worker.py", line 223, in determine_num_available_blocks
|
||||||
|
[1;36m(VllmWorkerProcess pid=6760)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] self.model_runner.profile_run()
|
||||||
|
[1;36m(VllmWorkerProcess pid=6760)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] File "/usr/local/corex/lib/python3/dist-packages/vllm/worker/multi_step_model_runner.py", line 661, in profile_run
|
||||||
|
[1;36m(VllmWorkerProcess pid=6760)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] return self._base_model_runner.profile_run()
|
||||||
|
[1;36m(VllmWorkerProcess pid=6760)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] File "/usr/local/corex/lib/python3/dist-packages/torch/utils/_contextlib.py", line 115, in decorate_context
|
||||||
|
[1;36m(VllmWorkerProcess pid=6760)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] return func(*args, **kwargs)
|
||||||
|
[1;36m(VllmWorkerProcess pid=6760)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] File "/usr/local/corex/lib/python3/dist-packages/vllm/worker/model_runner.py", line 1314, in profile_run
|
||||||
|
[1;36m(VllmWorkerProcess pid=6760)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] self.execute_model(model_input, kv_caches, intermediate_tensors)
|
||||||
|
[1;36m(VllmWorkerProcess pid=6760)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] File "/usr/local/corex/lib/python3/dist-packages/torch/utils/_contextlib.py", line 115, in decorate_context
|
||||||
|
[1;36m(VllmWorkerProcess pid=6760)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] return func(*args, **kwargs)
|
||||||
|
[1;36m(VllmWorkerProcess pid=6760)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] File "/usr/local/corex/lib/python3/dist-packages/vllm/worker/model_runner.py", line 1641, in execute_model
|
||||||
|
[1;36m(VllmWorkerProcess pid=6760)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] self.attn_state.begin_forward(model_input)
|
||||||
|
[1;36m(VllmWorkerProcess pid=6760)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] File "/usr/local/corex/lib/python3/dist-packages/vllm/attention/backends/flashinfer.py", line 261, in begin_forward
|
||||||
|
[1;36m(VllmWorkerProcess pid=6760)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] model_input.attn_metadata.decode_wrapper = state._get_decode_wrapper()
|
||||||
|
[1;36m(VllmWorkerProcess pid=6760)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] File "/usr/local/corex/lib/python3/dist-packages/vllm/attention/backends/flashinfer.py", line 130, in _get_decode_wrapper
|
||||||
|
[1;36m(VllmWorkerProcess pid=6760)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] self._decode_wrapper = BatchDecodeWithPagedKVCacheWrapper(
|
||||||
|
[1;36m(VllmWorkerProcess pid=6760)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231] TypeError: 'NoneType' object is not callable
|
||||||
|
[1;36m(VllmWorkerProcess pid=6760)[0;0m ERROR 07-14 09:44:24 multiproc_worker_utils.py:231]
|
||||||
|
Traceback (most recent call last):
|
||||||
|
File "/usr/local/lib/python3.10/runpy.py", line 196, in _run_module_as_main
|
||||||
|
return _run_code(code, main_globals, None,
|
||||||
|
File "/usr/local/lib/python3.10/runpy.py", line 86, in _run_code
|
||||||
|
exec(code, run_globals)
|
||||||
|
File "/usr/local/corex/lib/python3/dist-packages/vllm/entrypoints/openai/api_server.py", line 595, in <module>
|
||||||
|
uvloop.run(run_server(args))
|
||||||
|
File "/usr/local/lib/python3.10/site-packages/uvloop/__init__.py", line 82, in run
|
||||||
|
return loop.run_until_complete(wrapper())
|
||||||
|
File "uvloop/loop.pyx", line 1518, in uvloop.loop.Loop.run_until_complete
|
||||||
|
File "/usr/local/lib/python3.10/site-packages/uvloop/__init__.py", line 61, in wrapper
|
||||||
|
return await main
|
||||||
|
File "/usr/local/corex/lib/python3/dist-packages/vllm/entrypoints/openai/api_server.py", line 562, in run_server
|
||||||
|
async with build_async_engine_client(args) as engine_client:
|
||||||
|
File "/usr/local/lib/python3.10/contextlib.py", line 199, in __aenter__
|
||||||
|
return await anext(self.gen)
|
||||||
|
File "/usr/local/corex/lib/python3/dist-packages/vllm/entrypoints/openai/api_server.py", line 108, in build_async_engine_client
|
||||||
|
async with build_async_engine_client_from_engine_args(
|
||||||
|
File "/usr/local/lib/python3.10/contextlib.py", line 199, in __aenter__
|
||||||
|
return await anext(self.gen)
|
||||||
|
File "/usr/local/corex/lib/python3/dist-packages/vllm/entrypoints/openai/api_server.py", line 142, in build_async_engine_client_from_engine_args
|
||||||
|
engine_client = await asyncio.get_running_loop().run_in_executor(
|
||||||
|
File "/usr/local/lib/python3.10/concurrent/futures/thread.py", line 58, in run
|
||||||
|
result = self.fn(*self.args, **self.kwargs)
|
||||||
|
File "/usr/local/corex/lib/python3/dist-packages/vllm/engine/async_llm_engine.py", line 674, in from_engine_args
|
||||||
|
engine = cls(
|
||||||
|
File "/usr/local/corex/lib/python3/dist-packages/vllm/engine/async_llm_engine.py", line 569, in __init__
|
||||||
|
self.engine = self._engine_class(*args, **kwargs)
|
||||||
|
File "/usr/local/corex/lib/python3/dist-packages/vllm/engine/async_llm_engine.py", line 265, in __init__
|
||||||
|
super().__init__(*args, **kwargs)
|
||||||
|
File "/usr/local/corex/lib/python3/dist-packages/vllm/engine/llm_engine.py", line 349, in __init__
|
||||||
|
self._initialize_kv_caches()
|
||||||
|
File "/usr/local/corex/lib/python3/dist-packages/vllm/engine/llm_engine.py", line 484, in _initialize_kv_caches
|
||||||
|
self.model_executor.determine_num_available_blocks())
|
||||||
|
File "/usr/local/corex/lib/python3/dist-packages/vllm/executor/distributed_gpu_executor.py", line 39, in determine_num_available_blocks
|
||||||
|
num_blocks = self._run_workers("determine_num_available_blocks", )
|
||||||
@@ -0,0 +1,72 @@
|
|||||||
|
/usr/local/corex/lib/python3/dist-packages/torch/cuda/__init__.py:51: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you.
|
||||||
|
import pynvml # type: ignore[import]
|
||||||
|
INFO 07-14 09:23:27 importing.py:10] Triton not installed; certain GPU-related functions will not be available.
|
||||||
|
2026-07-14 09:23:29.177777: I tensorflow/core/util/port.cc:110] oneDNN custom operations are on. You may see slightly different numerical results due to floating-point round-off errors from different computation orders. To turn them off, set the environment variable `TF_ENABLE_ONEDNN_OPTS=0`.
|
||||||
|
2026-07-14 09:23:29.228965: I tensorflow/core/platform/cpu_feature_guard.cc:182] This TensorFlow binary is optimized to use available CPU instructions in performance-critical operations.
|
||||||
|
To enable the following instructions: SSE3 SSE4.1 SSE4.2 AVX AVX2 AVX512F AVX512_VNNI AVX512_BF16 AVX_VNNI AMX_TILE AMX_INT8 AMX_BF16 FMA, in other operations, rebuild TensorFlow with the appropriate compiler flags.
|
||||||
|
WARNING:tensorflow:Deprecation warnings have been disabled. Set TF_ENABLE_DEPRECATION_WARNINGS=1 to re-enable them.
|
||||||
|
INFO 07-14 09:23:34 api_server.py:530] vLLM API server version 0.6.3
|
||||||
|
INFO 07-14 09:23:34 api_server.py:531] args: Namespace(host='0.0.0.0', port=1111, uvicorn_log_level='info', allow_credentials=False, allowed_origins=['*'], allowed_methods=['*'], allowed_headers=['*'], api_key=None, lora_modules=None, prompt_adapters=None, chat_template=None, response_role='assistant', ssl_keyfile=None, ssl_certfile=None, ssl_ca_certs=None, ssl_cert_reqs=0, root_path=None, middleware=[], return_tokens_as_token_ids=False, disable_frontend_multiprocessing=True, enable_auto_tool_choice=True, tool_call_parser='qwen3_coder', tool_parser_plugin='', reasoning_parser='qwen3', model='/root/public-storage/models/Qwen/Qwen3.6-35B-A3B', tokenizer=None, skip_tokenizer_init=False, revision=None, code_revision=None, tokenizer_revision=None, tokenizer_mode='auto', trust_remote_code=True, download_dir=None, load_format='auto', config_format='auto', dtype='auto', kv_cache_dtype='auto', quantization_param_path=None, max_model_len=4096, guided_decoding_backend='outlines', distributed_executor_backend=None, worker_use_ray=False, pipeline_parallel_size=1, tensor_parallel_size=4, max_parallel_loading_workers=None, ray_workers_use_nsight=False, block_size=16, enable_prefix_caching=True, disable_sliding_window=False, use_v2_block_manager=True, num_lookahead_slots=0, seed=0, swap_space=4, cpu_offload_gb=0, gpu_memory_utilization=0.95, num_gpu_blocks_override=None, max_num_batched_tokens=8192, max_num_seqs=2, max_logprobs=20, disable_log_stats=False, quantization=None, rope_scaling=None, rope_theta=None, enforce_eager=False, max_context_len_to_capture=None, max_seq_len_to_capture=4096, disable_custom_all_reduce=False, tokenizer_pool_size=0, tokenizer_pool_type='ray', tokenizer_pool_extra_config=None, limit_mm_per_prompt=None, mm_processor_kwargs=None, enable_lora=False, max_loras=1, max_lora_rank=16, lora_extra_vocab_size=256, lora_dtype='auto', long_lora_scaling_factors=None, max_cpu_loras=None, fully_sharded_loras=False, enable_prompt_adapter=False, max_prompt_adapters=1, max_prompt_adapter_token=0, device='auto', num_scheduler_steps=4, multi_step_stream_outputs=True, scheduler_delay_factor=0.0, enable_chunked_prefill=None, speculative_model=None, speculative_model_quantization=None, num_speculative_tokens=None, speculative_disable_mqa_scorer=False, speculative_draft_tensor_parallel_size=None, speculative_max_model_len=None, speculative_disable_by_batch_size=None, ngram_prompt_lookup_max=None, ngram_prompt_lookup_min=None, spec_decoding_acceptance_method='rejection_sampler', typical_acceptance_sampler_posterior_threshold=None, typical_acceptance_sampler_posterior_alpha=None, disable_logprobs_during_spec_decoding=None, model_loader_extra_config=None, ignore_patterns=[], preemption_mode=None, served_model_name=['llm'], qlora_adapter_name_or_path=None, otlp_traces_endpoint=None, collect_detailed_traces=None, disable_async_output_proc=False, override_neuron_config=None, scheduling_policy='fcfs', disable_log_requests=True, max_log_len=None, disable_fastapi_docs=False)
|
||||||
|
INFO 07-14 09:23:34 config.py:1670] Downcasting torch.float32 to torch.float16.
|
||||||
|
INFO 07-14 09:23:45 config.py:887] Defaulting to use mp for distributed inference
|
||||||
|
WARNING 07-14 09:23:45 config.py:380] To see benefits of async output processing, enable CUDA graph. Since, enforce-eager is enabled, async output processor cannot be used
|
||||||
|
INFO 07-14 09:23:45 llm_engine.py:237] Initializing an LLM engine (v0.6.3) with config: model='/root/public-storage/models/Qwen/Qwen3.6-35B-A3B', speculative_config=None, tokenizer='/root/public-storage/models/Qwen/Qwen3.6-35B-A3B', skip_tokenizer_init=False, tokenizer_mode=auto, revision=None, override_neuron_config=None, rope_scaling=None, rope_theta=None, tokenizer_revision=None, trust_remote_code=True, dtype=torch.float16, max_seq_len=4096, download_dir=None, load_format=LoadFormat.AUTO, tensor_parallel_size=4, pipeline_parallel_size=1, disable_custom_all_reduce=True, quantization=None, enforce_eager=True, kv_cache_dtype=auto, quantization_param_path=None, device_config=cuda, decoding_config=DecodingConfig(guided_decoding_backend='outlines'), observability_config=ObservabilityConfig(otlp_traces_endpoint=None, collect_model_forward_time=False, collect_model_execute_time=False), seed=0, served_model_name=llm, use_v2_block_manager=True, num_scheduler_steps=4, chunked_prefill_enabled=False multi_step_stream_outputs=True, enable_prefix_caching=True, use_async_output_proc=False, use_cached_outputs=False, mm_processor_kwargs=None)
|
||||||
|
INFO 07-14 09:23:46 custom_cache_manager.py:17] Setting Triton cache manager to: vllm.triton_utils.custom_cache_manager:CustomCacheManager
|
||||||
|
INFO 07-14 09:23:46 selector.py:266] Cannot use FlashAttention-2 backend because the vllm.vllm_flash_attn package is not found. Make sure that vllm_flash_attn was built and installed (on by default).
|
||||||
|
INFO 07-14 09:23:46 selector.py:115] Using XFormers backend.
|
||||||
|
WARNING 07-14 09:23:46 registry.py:205] `mm_limits` has already been set for model=/root/public-storage/models/Qwen/Qwen3.6-35B-A3B, and will be overwritten by the new values.
|
||||||
|
Traceback (most recent call last):
|
||||||
|
File "/usr/local/lib/python3.10/runpy.py", line 196, in _run_module_as_main
|
||||||
|
return _run_code(code, main_globals, None,
|
||||||
|
File "/usr/local/lib/python3.10/runpy.py", line 86, in _run_code
|
||||||
|
exec(code, run_globals)
|
||||||
|
File "/usr/local/corex/lib/python3/dist-packages/vllm/entrypoints/openai/api_server.py", line 595, in <module>
|
||||||
|
uvloop.run(run_server(args))
|
||||||
|
File "/usr/local/lib/python3.10/site-packages/uvloop/__init__.py", line 82, in run
|
||||||
|
return loop.run_until_complete(wrapper())
|
||||||
|
File "uvloop/loop.pyx", line 1518, in uvloop.loop.Loop.run_until_complete
|
||||||
|
File "/usr/local/lib/python3.10/site-packages/uvloop/__init__.py", line 61, in wrapper
|
||||||
|
return await main
|
||||||
|
File "/usr/local/corex/lib/python3/dist-packages/vllm/entrypoints/openai/api_server.py", line 562, in run_server
|
||||||
|
async with build_async_engine_client(args) as engine_client:
|
||||||
|
File "/usr/local/lib/python3.10/contextlib.py", line 199, in __aenter__
|
||||||
|
return await anext(self.gen)
|
||||||
|
File "/usr/local/corex/lib/python3/dist-packages/vllm/entrypoints/openai/api_server.py", line 108, in build_async_engine_client
|
||||||
|
async with build_async_engine_client_from_engine_args(
|
||||||
|
File "/usr/local/lib/python3.10/contextlib.py", line 199, in __aenter__
|
||||||
|
return await anext(self.gen)
|
||||||
|
File "/usr/local/corex/lib/python3/dist-packages/vllm/entrypoints/openai/api_server.py", line 142, in build_async_engine_client_from_engine_args
|
||||||
|
engine_client = await asyncio.get_running_loop().run_in_executor(
|
||||||
|
File "/usr/local/lib/python3.10/concurrent/futures/thread.py", line 58, in run
|
||||||
|
result = self.fn(*self.args, **self.kwargs)
|
||||||
|
File "/usr/local/corex/lib/python3/dist-packages/vllm/engine/async_llm_engine.py", line 674, in from_engine_args
|
||||||
|
engine = cls(
|
||||||
|
File "/usr/local/corex/lib/python3/dist-packages/vllm/engine/async_llm_engine.py", line 569, in __init__
|
||||||
|
self.engine = self._engine_class(*args, **kwargs)
|
||||||
|
File "/usr/local/corex/lib/python3/dist-packages/vllm/engine/async_llm_engine.py", line 265, in __init__
|
||||||
|
super().__init__(*args, **kwargs)
|
||||||
|
File "/usr/local/corex/lib/python3/dist-packages/vllm/engine/llm_engine.py", line 335, in __init__
|
||||||
|
self.model_executor = executor_class(
|
||||||
|
File "/usr/local/corex/lib/python3/dist-packages/vllm/executor/multiproc_gpu_executor.py", line 215, in __init__
|
||||||
|
super().__init__(*args, **kwargs)
|
||||||
|
File "/usr/local/corex/lib/python3/dist-packages/vllm/executor/distributed_gpu_executor.py", line 26, in __init__
|
||||||
|
super().__init__(*args, **kwargs)
|
||||||
|
File "/usr/local/corex/lib/python3/dist-packages/vllm/executor/executor_base.py", line 47, in __init__
|
||||||
|
self._init_executor()
|
||||||
|
File "/usr/local/corex/lib/python3/dist-packages/vllm/executor/multiproc_gpu_executor.py", line 108, in _init_executor
|
||||||
|
self.driver_worker = self._create_worker(
|
||||||
|
File "/usr/local/corex/lib/python3/dist-packages/vllm/executor/gpu_executor.py", line 105, in _create_worker
|
||||||
|
return create_worker(**self._get_create_worker_kwargs(
|
||||||
|
File "/usr/local/corex/lib/python3/dist-packages/vllm/executor/gpu_executor.py", line 24, in create_worker
|
||||||
|
wrapper.init_worker(**kwargs)
|
||||||
|
File "/usr/local/corex/lib/python3/dist-packages/vllm/worker/worker_base.py", line 449, in init_worker
|
||||||
|
self.worker = worker_class(*args, **kwargs)
|
||||||
|
File "/usr/local/corex/lib/python3/dist-packages/vllm/worker/multi_step_worker.py", line 28, in __init__
|
||||||
|
self.model_runner = MultiStepModelRunner(
|
||||||
|
File "/usr/local/corex/lib/python3/dist-packages/vllm/worker/multi_step_model_runner.py", line 317, in __init__
|
||||||
|
raise ValueError(
|
||||||
|
ValueError: Multi-Step not supported for attention backend: xformers. Set VLLM_ATTENTION_BACKEND to a value from ['flash-attn', 'rocm-flash-attn', 'flashinfer'].
|
||||||
|
ERROR 07-14 09:23:46 multiproc_worker_utils.py:117] Worker VllmWorkerProcess pid 6686 died, exit code: -15
|
||||||
|
ERROR 07-14 09:23:46 multiproc_worker_utils.py:117] Worker VllmWorkerProcess pid 6687 died, exit code: -15
|
||||||
|
ERROR 07-14 09:23:46 multiproc_worker_utils.py:117] Worker VllmWorkerProcess pid 6688 died, exit code: -15
|
||||||
|
INFO 07-14 09:23:46 multiproc_worker_utils.py:121] Killing local vLLM worker processes
|
||||||
135
worklogs/remote_smoke_bench.py
Normal file
135
worklogs/remote_smoke_bench.py
Normal file
@@ -0,0 +1,135 @@
|
|||||||
|
import json
|
||||||
|
import statistics
|
||||||
|
import sys
|
||||||
|
import time
|
||||||
|
import urllib.request
|
||||||
|
from datetime import datetime
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
|
||||||
|
BASE_URL = sys.argv[1] if len(sys.argv) > 1 else "http://127.0.0.1:1111"
|
||||||
|
OUT_DIR = Path(sys.argv[2]) if len(sys.argv) > 2 else Path("/root/work/logs")
|
||||||
|
MODEL = sys.argv[3] if len(sys.argv) > 3 else "llm"
|
||||||
|
|
||||||
|
|
||||||
|
def post_json(path, payload, timeout=300):
|
||||||
|
req = urllib.request.Request(
|
||||||
|
BASE_URL + path,
|
||||||
|
data=json.dumps(payload, ensure_ascii=False).encode("utf-8"),
|
||||||
|
headers={"Content-Type": "application/json"},
|
||||||
|
method="POST",
|
||||||
|
)
|
||||||
|
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
||||||
|
return resp.status, resp.read().decode("utf-8")
|
||||||
|
|
||||||
|
|
||||||
|
def health():
|
||||||
|
with urllib.request.urlopen(BASE_URL + "/health", timeout=10) as resp:
|
||||||
|
return resp.status
|
||||||
|
|
||||||
|
|
||||||
|
def chat_once(prompt, max_tokens=32):
|
||||||
|
payload = {
|
||||||
|
"model": MODEL,
|
||||||
|
"messages": [{"role": "user", "content": prompt}],
|
||||||
|
"max_tokens": max_tokens,
|
||||||
|
"temperature": 0,
|
||||||
|
}
|
||||||
|
start = time.perf_counter()
|
||||||
|
status, body = post_json("/v1/chat/completions", payload)
|
||||||
|
elapsed = time.perf_counter() - start
|
||||||
|
obj = json.loads(body)
|
||||||
|
usage = obj.get("usage") or {}
|
||||||
|
completion_tokens = usage.get("completion_tokens") or 0
|
||||||
|
return {
|
||||||
|
"status": status,
|
||||||
|
"elapsed_sec": elapsed,
|
||||||
|
"usage": usage,
|
||||||
|
"output_tps": completion_tokens / elapsed if elapsed and completion_tokens else None,
|
||||||
|
"content_preview": (obj["choices"][0]["message"].get("content") or "")[:200],
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def chat_stream(prompt, max_tokens=64):
|
||||||
|
payload = {
|
||||||
|
"model": MODEL,
|
||||||
|
"messages": [{"role": "user", "content": prompt}],
|
||||||
|
"max_tokens": max_tokens,
|
||||||
|
"temperature": 0,
|
||||||
|
"stream": True,
|
||||||
|
"stream_options": {"include_usage": True},
|
||||||
|
}
|
||||||
|
req = urllib.request.Request(
|
||||||
|
BASE_URL + "/v1/chat/completions",
|
||||||
|
data=json.dumps(payload, ensure_ascii=False).encode("utf-8"),
|
||||||
|
headers={"Content-Type": "application/json"},
|
||||||
|
method="POST",
|
||||||
|
)
|
||||||
|
start = time.perf_counter()
|
||||||
|
first_token_at = None
|
||||||
|
usage = None
|
||||||
|
pieces = []
|
||||||
|
with urllib.request.urlopen(req, timeout=300) as resp:
|
||||||
|
for raw in resp:
|
||||||
|
line = raw.decode("utf-8", errors="replace").strip()
|
||||||
|
if not line or not line.startswith("data:"):
|
||||||
|
continue
|
||||||
|
data = line[5:].strip()
|
||||||
|
if data == "[DONE]":
|
||||||
|
break
|
||||||
|
obj = json.loads(data)
|
||||||
|
if obj.get("usage"):
|
||||||
|
usage = obj["usage"]
|
||||||
|
choices = obj.get("choices") or []
|
||||||
|
if choices:
|
||||||
|
delta = choices[0].get("delta") or {}
|
||||||
|
text = delta.get("content") or delta.get("reasoning_content") or ""
|
||||||
|
if text:
|
||||||
|
if first_token_at is None:
|
||||||
|
first_token_at = time.perf_counter()
|
||||||
|
pieces.append(text)
|
||||||
|
elapsed = time.perf_counter() - start
|
||||||
|
completion_tokens = (usage or {}).get("completion_tokens") or 0
|
||||||
|
gen_time = elapsed - (first_token_at - start) if first_token_at else elapsed
|
||||||
|
return {
|
||||||
|
"elapsed_sec": elapsed,
|
||||||
|
"ttft_sec": (first_token_at - start) if first_token_at else None,
|
||||||
|
"usage": usage,
|
||||||
|
"output_tps_after_ttft": completion_tokens / gen_time if gen_time and completion_tokens else None,
|
||||||
|
"content_preview": "".join(pieces)[:200],
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
OUT_DIR.mkdir(parents=True, exist_ok=True)
|
||||||
|
results = {
|
||||||
|
"base_url": BASE_URL,
|
||||||
|
"model": MODEL,
|
||||||
|
"created_at": datetime.now().isoformat(timespec="seconds"),
|
||||||
|
}
|
||||||
|
results["health_status"] = health()
|
||||||
|
results["smoke_nonstream"] = chat_once("你好,请用一句话介绍你自己。", max_tokens=16)
|
||||||
|
|
||||||
|
short_prompt = "请用中文简要说明什么是模型推理服务。"
|
||||||
|
stream_runs = [chat_stream(short_prompt, max_tokens=64) for _ in range(3)]
|
||||||
|
results["stream_runs"] = stream_runs
|
||||||
|
ttfts = [x["ttft_sec"] for x in stream_runs if x["ttft_sec"] is not None]
|
||||||
|
tps = [x["output_tps_after_ttft"] for x in stream_runs if x["output_tps_after_ttft"] is not None]
|
||||||
|
results["stream_summary"] = {
|
||||||
|
"ttft_avg_sec": statistics.mean(ttfts) if ttfts else None,
|
||||||
|
"ttft_p90_sec": sorted(ttfts)[int(0.9 * (len(ttfts) - 1))] if ttfts else None,
|
||||||
|
"output_tps_avg": statistics.mean(tps) if tps else None,
|
||||||
|
}
|
||||||
|
|
||||||
|
repeated = "以下是一段用于测试前缀缓存的公共上下文:" + ("模型部署竞赛关注吞吐、延迟、缓存命中和稳定性。" * 80)
|
||||||
|
results["prefix_cache_probe_1"] = chat_once(repeated + "\n请总结一句话。", max_tokens=16)
|
||||||
|
results["prefix_cache_probe_2"] = chat_once(repeated + "\n请换一种说法总结一句话。", max_tokens=16)
|
||||||
|
|
||||||
|
out = OUT_DIR / f"baseline_smoke_{datetime.now().strftime('%Y%m%d_%H%M%S')}.json"
|
||||||
|
out.write_text(json.dumps(results, ensure_ascii=False, indent=2), encoding="utf-8")
|
||||||
|
print(json.dumps(results, ensure_ascii=False, indent=2), flush=True)
|
||||||
|
print("RESULT_FILE", out, flush=True)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
Reference in New Issue
Block a user