Skip to content

Commit a61da71

Browse files
authored
[Build] Build tilelang without host toolchain (#1833)
Build tilelang without host toolchain
1 parent 7ab2b2b commit a61da71

9 files changed

Lines changed: 218 additions & 77 deletions

File tree

‎.gitignore‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -102,6 +102,7 @@ tilelang/jit/adapter/cython/.cycache
102102
# CMake
103103
cmake-build/
104104
cmake-build-*/
105+
CMakeFiles/
105106

106107
# Git version for sdist
107108
.git_commit.txt

‎CMakeLists.txt‎

Lines changed: 7 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -2,6 +2,12 @@
22
# https://github.com/mlc-ai/mlc-llm/blob/main/CMakeLists.txt
33

44
cmake_minimum_required(VERSION 3.26)
5+
6+
# Detect CUDA toolkit: tries host installation first, then falls back to
7+
# pip-installed packages (env WITH_PIP_CUDA_TOOLCHAIN or auto-detect).
8+
# Must be included before project() so CMAKE_CUDA_COMPILER is set.
9+
include(${CMAKE_CURRENT_LIST_DIR}/cmake/FindPipCUDAToolkit.cmake)
10+
511
project(TILE_LANG C CXX)
612

713
set(CMAKE_CXX_STANDARD 17)
@@ -334,6 +340,7 @@ elseif(USE_CUDA)
334340
list(APPEND TILE_LANG_SRCS ${TILE_LANG_CUDA_SRCS})
335341

336342
list(APPEND TILE_LANG_INCLUDES ${CUDAToolkit_INCLUDE_DIRS})
343+
link_directories(${CUDAToolkit_LIBRARY_DIR} ${CUDAToolkit_LIBRARY_DIR}/stubs)
337344
endif()
338345

339346
set(USE_Z3 ON CACHE STRING "Use Z3 SMT solver for TileLang optimizations")

‎cmake/FindPipCUDAToolkit.cmake‎

Lines changed: 70 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,70 @@
1+
# FindPipCUDAToolkit.cmake
2+
#
3+
# Locate CUDA toolkit — first trying the host system, then falling back
4+
# to pip-installed packages (nvidia-cuda-nvcc, nvidia-cuda-cccl).
5+
#
6+
# This module should be included BEFORE project() to set CMAKE_CUDA_COMPILER
7+
# when pip CUDA is used.
8+
#
9+
# Detection order:
10+
# 1. Try find_package(CUDAToolkit QUIET) — succeeds if a host CUDA
11+
# installation is available; skip pip detection.
12+
# 2. If env var WITH_PIP_CUDA_TOOLCHAIN is set to a path (e.g., .../cu13),
13+
# use that directory directly as the CUDA toolkit root.
14+
# 3. Otherwise, try auto-detecting from the current Python environment's
15+
# site-packages (works with --no-build-isolation).
16+
17+
# --- Try host CUDA first ---
18+
find_package(CUDAToolkit QUIET)
19+
if(CUDAToolkit_FOUND)
20+
return()
21+
endif()
22+
23+
find_program(_PIP_CUDA_PYTHON_EXE NAMES python3 python)
24+
if(NOT _PIP_CUDA_PYTHON_EXE)
25+
return()
26+
endif()
27+
28+
# --- Strategy 1: explicit path via env var ---
29+
if(DEFINED ENV{WITH_PIP_CUDA_TOOLCHAIN})
30+
set(_PIP_CUDA_ROOT "$ENV{WITH_PIP_CUDA_TOOLCHAIN}")
31+
if(NOT EXISTS "${_PIP_CUDA_ROOT}/bin/nvcc")
32+
message(FATAL_ERROR
33+
"FindPipCUDAToolkit: WITH_PIP_CUDA_TOOLCHAIN is set to '${_PIP_CUDA_ROOT}' "
34+
"but nvcc was not found at '${_PIP_CUDA_ROOT}/bin/nvcc'")
35+
endif()
36+
# Prepare the directory (create lib64 symlink, unversioned .so symlinks,
37+
# libcuda.so stub) that CMake / nvcc expect but pip packages omit.
38+
execute_process(
39+
COMMAND "${_PIP_CUDA_PYTHON_EXE}" "${CMAKE_CURRENT_LIST_DIR}/find_pip_cuda.py"
40+
"${_PIP_CUDA_ROOT}"
41+
OUTPUT_QUIET
42+
)
43+
message(STATUS "FindPipCUDAToolkit: using env WITH_PIP_CUDA_TOOLCHAIN=${_PIP_CUDA_ROOT}")
44+
else()
45+
# --- Strategy 2: auto-detect from current Python env ---
46+
execute_process(
47+
COMMAND "${_PIP_CUDA_PYTHON_EXE}" "${CMAKE_CURRENT_LIST_DIR}/find_pip_cuda.py"
48+
OUTPUT_VARIABLE _PIP_CUDA_OUTPUT
49+
OUTPUT_STRIP_TRAILING_WHITESPACE
50+
RESULT_VARIABLE _PIP_CUDA_RESULT
51+
)
52+
53+
if(NOT _PIP_CUDA_RESULT EQUAL 0)
54+
message(STATUS "FindPipCUDAToolkit: pip-installed CUDA toolkit not found")
55+
return()
56+
endif()
57+
58+
string(JSON _PIP_CUDA_ROOT GET "${_PIP_CUDA_OUTPUT}" "root")
59+
message(STATUS "FindPipCUDAToolkit: auto-detected from Python environment")
60+
endif()
61+
62+
# --- Common pip-CUDA setup ---
63+
set(CMAKE_CUDA_COMPILER "${_PIP_CUDA_ROOT}/bin/nvcc" CACHE FILEPATH "CUDA compiler (from pip)" FORCE)
64+
set(CUDAToolkit_ROOT "${_PIP_CUDA_ROOT}" CACHE PATH "CUDA toolkit root (from pip)" FORCE)
65+
66+
list(APPEND CMAKE_LIBRARY_PATH "${_PIP_CUDA_ROOT}/lib/stubs" "${_PIP_CUDA_ROOT}/lib")
67+
68+
message(STATUS "FindPipCUDAToolkit: using pip-installed CUDA toolkit")
69+
message(STATUS " nvcc: ${CMAKE_CUDA_COMPILER}")
70+
message(STATUS " root: ${CUDAToolkit_ROOT}")

‎cmake/find_pip_cuda.py‎

Lines changed: 103 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,103 @@
1+
"""Locate pip-installed CUDA toolkit and prepare it for CMake consumption.
2+
3+
Used by cmake/FindPipCUDAToolkit.cmake via ``execute_process``.
4+
Outputs a JSON object with paths on success, exits with code 1 on failure.
5+
6+
Usage:
7+
python find_pip_cuda.py # auto-detect from current env
8+
python find_pip_cuda.py /path/to/cu13 # use explicit path, just prepare it
9+
"""
10+
11+
import contextlib
12+
import json
13+
import pathlib
14+
import subprocess
15+
import sys
16+
17+
18+
def _find_cu_dir():
19+
"""Find the nvidia/cu<ver> directory from the nvidia pip package."""
20+
try:
21+
import nvidia
22+
except ImportError:
23+
return None
24+
25+
nvidia_dir = pathlib.Path(nvidia.__path__[0])
26+
cu_dirs = sorted(
27+
(d for d in nvidia_dir.iterdir() if d.name[:2] == "cu" and d.name[2:].isdigit()),
28+
key=lambda d: int(d.name[2:]),
29+
)
30+
if not cu_dirs:
31+
return None
32+
cu_dir = cu_dirs[-1]
33+
if (cu_dir / "bin" / "nvcc").is_file():
34+
return cu_dir
35+
return None
36+
37+
38+
def _ensure_lib_symlinks(cu_dir):
39+
"""Create symlinks that CMake / nvcc expect but pip packages omit."""
40+
lib_dir = cu_dir / "lib"
41+
if not lib_dir.is_dir():
42+
return
43+
44+
# nvcc expects lib64/ on 64-bit
45+
lib64 = cu_dir / "lib64"
46+
if not lib64.exists():
47+
with contextlib.suppress(OSError):
48+
lib64.symlink_to("lib")
49+
50+
# CMake expects unversioned .so (e.g., libcudart.so)
51+
for so in lib_dir.glob("*.so.*"):
52+
base = lib_dir / (so.name.split(".so.")[0] + ".so")
53+
if not base.exists():
54+
with contextlib.suppress(OSError):
55+
base.symlink_to(so.name)
56+
57+
58+
def _ensure_cuda_stub(cu_dir):
59+
"""Create a minimal libcuda.so stub for build-time -lcuda linking."""
60+
stubs_dir = cu_dir / "lib" / "stubs"
61+
stub = stubs_dir / "libcuda.so"
62+
if stub.exists():
63+
return
64+
stubs_dir.mkdir(parents=True, exist_ok=True)
65+
src = stubs_dir / "_stub.c"
66+
try:
67+
src.write_text("void cuGetErrorString(void){}\n")
68+
subprocess.check_call(
69+
["gcc", "-shared", "-o", str(stub), str(src)],
70+
stderr=subprocess.DEVNULL,
71+
)
72+
except Exception:
73+
pass
74+
finally:
75+
src.unlink(missing_ok=True)
76+
77+
78+
def main():
79+
if len(sys.argv) > 1:
80+
# Explicit path provided — just prepare it
81+
cu_dir = pathlib.Path(sys.argv[1])
82+
else:
83+
# Auto-detect from current Python environment
84+
cu_dir = _find_cu_dir()
85+
86+
if cu_dir is None or not (cu_dir / "bin" / "nvcc").is_file():
87+
sys.exit(1)
88+
89+
_ensure_lib_symlinks(cu_dir)
90+
_ensure_cuda_stub(cu_dir)
91+
92+
print(
93+
json.dumps(
94+
{
95+
"nvcc": str(cu_dir / "bin" / "nvcc"),
96+
"root": str(cu_dir),
97+
}
98+
)
99+
)
100+
101+
102+
if __name__ == "__main__":
103+
main()

‎docs/get_started/Installation.md‎

Lines changed: 30 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -5,8 +5,8 @@
55
**Prerequisites for installation via wheel or PyPI:**
66

77
- **glibc**: 2.28 (Ubuntu 20.04 or later)
8-
- **Python Version**: >= 3.8
9-
- **CUDA Version**: 12.0 <= CUDA < 13
8+
- **Python Version**: >= 3.9
9+
- **CUDA Version**: >= 10.0 (host installation), or pip-provided CUDA toolchain (>= 13.0)
1010

1111
The easiest way to install tilelang is directly from PyPI using pip. To install the latest version, run the following command in your terminal:
1212

@@ -37,8 +37,8 @@ python -c "import tilelang; print(tilelang.__version__)"
3737
**Prerequisites for building from source:**
3838

3939
- **Operating System**: Linux
40-
- **Python Version**: >= 3.8
41-
- **CUDA Version**: >= 10.0
40+
- **Python Version**: >= 3.9
41+
- **CUDA Version**: >= 10.0 (host installation), or pip-provided CUDA toolchain (>= 13.0)
4242

4343
If you prefer Docker, please skip to the [Install Using Docker](#install-using-docker) section. This section focuses on building from source on a native Linux environment.
4444

@@ -53,9 +53,33 @@ Then, clone the tilelang repository and install it using pip. The `-v` flag enab
5353

5454
> **Note**: Use the `--recursive` flag to include necessary submodules. Tilelang currently depends on a customized version of TVM, which is included as a submodule. If you prefer [Building with Existing TVM Installation](#using-existing-tvm), you can skip cloning the TVM submodule (but still need other dependencies).
5555
56+
### With host CUDA toolchain
57+
58+
```bash
59+
git clone --recursive https://github.com/tile-ai/tilelang.git
60+
cd tilelang
61+
pip install . -v
62+
```
63+
64+
### With pip-provided CUDA toolchain (no host CUDA required)
65+
66+
If you don't have CUDA installed on the host, you can use pip-provided CUDA packages instead.
67+
68+
**Option A** — pip toolchain in the current environment (use `--no-build-isolation`):
69+
5670
```bash
5771
git clone --recursive https://github.com/tile-ai/tilelang.git
5872
cd tilelang
73+
pip install -r requirements-dev.txt
74+
pip install "nvidia-cuda-nvcc>=13" "nvidia-cuda-cccl>=13" "nvidia-cuda-nvrtc>=13"
75+
pip install . -v --no-build-isolation
76+
```
77+
78+
**Option B** — pip toolchain in another virtualenv or path:
79+
80+
```bash
81+
# Point to the cu<ver> directory inside another venv's site-packages
82+
export WITH_PIP_CUDA_TOOLCHAIN=/path/to/venv/lib/python3.x/site-packages/nvidia/cu13
5983
pip install . -v
6084
```
6185

@@ -281,6 +305,8 @@ pip install tilelang -f https://tile-ai.github.io/whl/nightly
281305

282306
`TVM_ROOT`: TVM source root to use.
283307

308+
`WITH_PIP_CUDA_TOOLCHAIN`: Path to a pip-installed CUDA toolkit directory (e.g., `/path/to/venv/lib/python3.x/site-packages/nvidia/cu13`). When set, the build system uses this directory instead of a host CUDA installation. If not set and no host CUDA is found, the build system will attempt to auto-detect pip-installed CUDA packages from the current Python environment.
309+
284310
`NO_VERSION_LABEL` and `NO_TOOLCHAIN_VERSION`:
285311
When building tilelang, we'll try to embed SDK and version information into package version as below,
286312
where local version label could look like `<sdk>.git<git_hash>`. Set `NO_VERSION_LABEL=ON` to disable this behavior.

‎requirements-dev.txt‎

Lines changed: 3 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -1,17 +1,16 @@
11
# Requirements to run local build with `--no-build-isolation` or other developments
22

3-
apache-tvm-ffi~=0.1.0,>=0.1.6
3+
apache-tvm-ffi~=0.1.0,>=0.1.2
44
build
55
cmake>=3.26
6-
cython>=3.0.0
6+
cython>=3.1.0
77
ninja
88
packaging
99
pre-commit>=4.0.0
1010
scikit-build-core
1111
setuptools>=61
12-
torch
1312
wheel
14-
z3-solver>=4.13.0
13+
z3-solver>=4.13.0,<4.15.5
1514

1615
auditwheel; platform_system == 'Linux'
1716
patchelf; platform_system == 'Linux'

‎requirements-test.txt‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -33,4 +33,4 @@ seaborn
3333
tabulate
3434
tornado
3535
wheel
36-
z3-solver>=4.13.0
36+
z3-solver>=4.13.0,<4.15.5

‎requirements.txt‎

Lines changed: 3 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -1,7 +1,7 @@
11
# Runtime requirements
22

3-
apache-tvm-ffi~=0.1.0,>=0.1.6
4-
torch-c-dlpack-ext
3+
apache-tvm-ffi~=0.1.0,>=0.1.2
4+
torch-c-dlpack-ext; python_version < '3.14'
55
cloudpickle
66
ml-dtypes
77
numpy>=1.23.5
@@ -10,4 +10,4 @@ torch
1010
torch>=2.7; platform_system == 'Darwin'
1111
tqdm>=4.62.3
1212
typing-extensions>=4.10.0
13-
z3-solver>=4.13.0
13+
z3-solver>=4.13.0,<4.15.5

‎tilelang/contrib/nvcc.py‎

Lines changed: 0 additions & 65 deletions
Original file line numberDiff line numberDiff line change
@@ -319,71 +319,6 @@ def get_cuda_version(cuda_path=None):
319319
raise RuntimeError("Cannot read cuda version file")
320320

321321

322-
@tvm_ffi.register_global_func("tilelang_callback_libdevice_path", override=True)
323-
def find_libdevice_path(arch):
324-
"""Utility function to find libdevice
325-
326-
Parameters
327-
----------
328-
arch : int
329-
The compute architecture in int
330-
331-
Returns
332-
-------
333-
path : str
334-
Path to libdevice.
335-
"""
336-
cuda_path = find_cuda_path()
337-
lib_path = os.path.join(cuda_path, "nvvm/libdevice")
338-
if not os.path.exists(lib_path):
339-
# Debian/Ubuntu repackaged CUDA path
340-
lib_path = os.path.join(cuda_path, "lib/nvidia-cuda-toolkit/libdevice")
341-
selected_ver = 0
342-
selected_path = None
343-
cuda_ver = get_cuda_version(cuda_path)
344-
major_minor = (cuda_ver[0], cuda_ver[1])
345-
if major_minor in (
346-
(9, 0),
347-
(9, 1),
348-
(10, 0),
349-
(10, 1),
350-
(10, 2),
351-
(11, 0),
352-
(11, 1),
353-
(11, 2),
354-
(11, 3),
355-
):
356-
path = os.path.join(lib_path, "libdevice.10.bc")
357-
else:
358-
for fn in os.listdir(lib_path):
359-
if not fn.startswith("libdevice"):
360-
continue
361-
362-
try:
363-
# expected pattern: libdevice.${ARCH}.10.bc
364-
# e.g., libdevice.compute_20.10.bc
365-
ver = int(fn.split(".")[-3].split("_")[-1])
366-
if selected_ver < ver <= arch:
367-
selected_ver = ver
368-
selected_path = fn
369-
except ValueError:
370-
# it can just be `libdevice.10.bc` in CUDA 10
371-
selected_path = fn
372-
373-
if selected_path is None:
374-
raise RuntimeError(f"Cannot find libdevice for arch {arch}")
375-
path = os.path.join(lib_path, selected_path)
376-
return path
377-
378-
379-
def callback_libdevice_path(arch):
380-
try:
381-
return find_libdevice_path(arch)
382-
except RuntimeError:
383-
warnings.warn("Cannot find libdevice path", stacklevel=2)
384-
return ""
385-
386-
387322
@tvm_ffi.register_global_func("tvm.contrib.nvcc.get_compute_version", override=True)
388323
def get_target_compute_version(target=None):
389324
"""Utility function to get compute capability of compilation target.

0 commit comments

Comments
 (0)