Files
project_6_89d52222/upstream_ref/xllm/xllm/pybind/multimodal.py
EX Engine 002f9879b2 ref(upstream): FULL TREE — Deep-Spark xllm (1470) + ds_vllm csrc/models (703)
Replaces cherry-picked upstream_ref with complete source trees.

xllm/ — Iluvatar official C++ inference engine (15MB, 1470 files)
  Complete: kernels → layers → models → runtime → scheduler → api
  Excluded: .git, binary images, third_party submodule checkouts

ds_vllm/ — Iluvatar official vllm fork (8MB, 703 files)
  Included: csrc/ (ALL CUDA kernels), fused_moe/, qwen3_5 model, _custom_ops
  Excluded: tests, benchmarks, docs, examples (not needed for reference)

Critical call chains now fully traceable:
  MoE: moe_topk_softmax_kernels.cuh → ixformer.h → fused_moe.cpp → layer
  GDN: qwen3_gated_delta_net_base.cpp → qwen3_5_gated_delta_net.cpp
  Attention: ixformer.h → xllm_paged_attention → attention.cpp
2026-08-10 02:54:03 +00:00

71 lines
2.2 KiB
Python

from typing import Any, Dict, List, cast
from functools import lru_cache
from io import BytesIO
from PIL import Image
import torch
@lru_cache(maxsize=1)
def __cache_image_processor(
processor_name: str,
*args: Any,
trust_remote_code: bool = False,
**kwargs: Any,
):
"""Load an image processor for the given model name via HuggingFace."""
# don't put this import at the top level
# it will call torch.cuda.device_count()
from transformers import AutoImageProcessor
from transformers.image_processing_utils import BaseImageProcessor
try:
processor = AutoImageProcessor.from_pretrained(
processor_name,
*args,
trust_remote_code=trust_remote_code,
**kwargs)
except ValueError as e:
if not trust_remote_code:
err_msg = (
"Failed to load the image processor. If the image processor is "
"a custom processor not yet available in the HuggingFace "
"transformers library, consider setting "
"`trust_remote_code=True` in LLM or using the "
"`--trust-remote-code` flag in the CLI.")
raise RuntimeError(err_msg) from e
else:
raise e
return cast(BaseImageProcessor, processor)
def try_cat_feature(item):
if isinstance(item, torch.Tensor):
return item
assert isinstance(item, list), f"expected list, got {type(item)}"
lst = []
for i in item:
res = try_cat_feature(i)
if isinstance(res, list):
lst.extend(res)
elif isinstance(res, torch.Tensor):
lst.append(res)
else:
raise TypeError(f"expected list or torch.Tensor, got {type(res)}")
if len(lst) == 1:
return lst[0]
if any(t.shape[1:] != lst[0].shape[1:] for t in lst):
return lst
return torch.cat(lst)
def preprocess(lst: List[str], model: str) -> Dict[str, Any]:
images = [Image.open(BytesIO(item)) for item in lst]
image_processor = __cache_image_processor(model, trust_remote_code=True)
data = image_processor.preprocess(images, return_tensors="pt").data
return { key: try_cat_feature(val)
for key, val in data.items()
}