build(SM70): precompile GDN CUDA kernel to .so during docker build

precompile_gdn.py: calls torch.utils.cpp_extension.load with build_directory
to produce .so at build time. If build env has no GPU/compiler, fails
gracefully — kernel JIT compiles at runtime instead.

fused_fwd.py: _load_ext() now checks build/ dir for precompiled .so first,
skips 2-minute JIT compilation if found.
This commit is contained in:
Claude
2026-08-10 01:08:38 +00:00
parent 8cf73ad39c
commit 20cd2d8904
3 changed files with 75 additions and 1 deletions

View File

@@ -20,6 +20,24 @@ def _load_ext():
raise RuntimeError("SM70 FlashQLA backend requires CUDA.")
os.environ.setdefault("TORCH_CUDA_ARCH_LIST", "7.0;7.5")
# Try precompiled .so first (built during docker build)
build_dir = Path(__file__).with_name("build")
if build_dir.is_dir():
so_files = list(build_dir.glob("*.so"))
if so_files:
try:
_EXT = load(
name="flash_qla_sm70_gdn_strided",
sources=[], # empty — just load from build_directory
build_directory=str(build_dir),
verbose=False,
)
return _EXT
except Exception:
pass # fall through to JIT
# JIT compile (slow, ~2min first time)
src = Path(__file__).with_name("csrc") / "gdn_forward.cu"
_EXT = load(
name="flash_qla_sm70_gdn_strided",