Skip to content

Reject unsupported atomic_addx4 destination dtypes #8371

Reject unsupported atomic_addx4 destination dtypes

Reject unsupported atomic_addx4 destination dtypes #8371

Workflow file for this run

name: CI
on:
pull_request:
types:
- labeled
- unlabeled
- opened
- synchronize
- reopened
- ready_for_review
# Allow to trigger the workflow manually
workflow_dispatch:
permissions:
contents: read
concurrency:
group: "${{ github.workflow }}-${{ github.ref }}"
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
env:
PYTHONDEVMODE: "1"
PYTHONUNBUFFERED: "1"
PYTHONPATH: "" # explicit cleanup
PIP_USER: "" # explicit cleanup
COLUMNS: "100"
FORCE_COLOR: "1"
CLICOLOR_FORCE: "1"
UV_INDEX_STRATEGY: "unsafe-best-match"
UV_HTTP_TIMEOUT: "600"
XDG_CACHE_HOME: "${{ github.workspace }}/.cache" # to be updated
PIP_CACHE_DIR: "${{ github.workspace }}/.cache/pip" # to be updated
UV_CACHE_DIR: "${{ github.workspace }}/.cache/uv" # to be updated
PRE_COMMIT_HOME: "${{ github.workspace }}/.cache/pip/.pre-commit" # to be updated
jobs:
lint:
name: Quick Lint
runs-on: ubuntu-latest
timeout-minutes: 30
steps:
- name: Checkout repository
uses: actions/checkout@v7
with:
fetch-depth: 0
submodules: recursive
- name: Setup Python 3.10
id: setup-pylowest
uses: actions/setup-python@v7
with:
python-version: "3.10"
update-environment: true
cache: pip
cache-dependency-path: |
pyproject.toml
requirements*.txt
.pre-commit-config.yaml
- name: Check AST with Python 3.10
run: |
"${{ steps.setup-pylowest.outputs.python-path }}" -m compileall -q -f tilelang
- name: C++ API Style Audit (warning only)
env:
PYTHONDONTWRITEBYTECODE: "1"
run: |
"${{ steps.setup-pylowest.outputs.python-path }}" maint/scripts/audit_cpp_api_style.py --github-warnings --limit 20
- name: Pre-commit Lint
run: |
if ! pipx run pre-commit run --all-files --color=always --show-diff-on-failure; then
echo "::error::Pre-commit checks failed. Please run 'pre-commit install' and 'pre-commit run --all-files' locally to see the issues."
exit 1
fi
tests:
name: Test for Python ${{ matrix.python-version }} with ${{ matrix.runner.toolkit }} (on ${{ matrix.runner.name }})
if: |
github.repository_owner == 'tile-ai' &&
(github.event_name != 'pull_request' || !github.event.pull_request.draft)
needs: [lint]
runs-on: ${{ matrix.runner.tags }}
strategy:
matrix:
runner:
- tags: [self-hosted, nvidia]
name: self-hosted-nvidia
# Format: CUDA-auto or [Nightly-]CUDA-<major>.<minor>[.<patch>].
# E.g., "CUDA-auto", "CUDA-13.0", or "Nightly-CUDA-13.0".
# Use "Nightly-" prefix to use torch nightly builds.
toolkit: CUDA-auto
- tags: [self-hosted, amd, gfx942]
name: self-hosted-amd-gfx942
# Format: [Nightly-]ROCm-<major>.<minor>[.<patch>]. E.g., "ROCm-6.4" or "Nightly-ROCm-7.0".
# Use "Nightly-" prefix to use torch nightly builds.
toolkit: ROCm-7.2
- tags: [macos-latest]
name: macos-latest
toolkit: Metal # or Nightly-Metal
python-version:
- "3.12"
fail-fast: false
timeout-minutes: 120
steps:
- name: Checkout repository
uses: actions/checkout@v7
with:
fetch-depth: 0
submodules: recursive
- name: Set environment (self-hosted runners)
if: startsWith(matrix.runner.name, 'self-hosted')
run: |
# Hide sensitive data in logs for self-hosted runners
if [[ -n "${{ secrets.SECRET_PATH_PREFIXES }}" ]]; then
echo "::add-mask::${{ secrets.SECRET_PATH_PREFIXES }}"
# Colon separated list of secrets to mask
for secret in $(echo "${{ secrets.SECRET_PATH_PREFIXES }}" | tr ':' '\n'); do
echo "::add-mask::${secret}"
done
fi
# Use runner tool_cache as cache root for self-hosted runners to avoid internet connection
# issues and to share cache between jobs.
export XDG_CACHE_HOME="${{ runner.tool_cache }}/.ci-cache-${{ github.workflow }}"
echo "XDG_CACHE_HOME=${XDG_CACHE_HOME}" | tee -a "${GITHUB_ENV}"
echo "PIP_CACHE_DIR=${XDG_CACHE_HOME}/pip" | tee -a "${GITHUB_ENV}"
echo "UV_CACHE_DIR=${XDG_CACHE_HOME}/uv" | tee -a "${GITHUB_ENV}"
echo "PRE_COMMIT_HOME=${XDG_CACHE_HOME}/pip/.pre-commit" | tee -a "${GITHUB_ENV}"
# Use the ccache action only on GitHub-hosted runners, where it manages
# remote cache restore/save. Self-hosted runners use a local cache below.
- name: Setup ccache (GitHub-hosted runners)
id: setup-ccache
if: ${{ !startsWith(matrix.runner.name, 'self-hosted') }}
uses: hendrikmuhs/ccache-action@v1
with:
create-symlink: true
evict-old-files: "7d"
append-timestamp: false
key: ${{ runner.os }}-${{ runner.arch }}-${{ matrix.runner.toolkit }}-${{ hashFiles('**/*.cc') }}
restore-keys: |
${{ runner.os }}-${{ runner.arch }}-${{ matrix.runner.toolkit }}-${{ hashFiles('**/*.cc') }}
${{ runner.os }}-${{ runner.arch }}-${{ matrix.runner.toolkit }}
${{ runner.os }}-${{ runner.arch }}
- name: Setup ccache (self-hosted runners)
if: startsWith(matrix.runner.name, 'self-hosted')
run: |
if ! command -v ccache >/dev/null 2>&1; then
echo "::warning::ccache is not installed on this self-hosted runner; CMake will build without ccache."
exit 0
fi
export CCACHE_DIR="${XDG_CACHE_HOME}/ccache/${{ matrix.runner.name }}/${{ matrix.runner.toolkit }}"
mkdir -p "${CCACHE_DIR}"
echo "CCACHE_DIR=${CCACHE_DIR}" | tee -a "${GITHUB_ENV}"
echo "CCACHE_BASEDIR=${{ github.workspace }}" | tee -a "${GITHUB_ENV}"
echo "CCACHE_NOHASHDIR=true" | tee -a "${GITHUB_ENV}"
echo "CCACHE_COMPILERCHECK=content" | tee -a "${GITHUB_ENV}"
ccache -M 20G
ccache -z
ccache -s
- name: Set environment (CUDA)
if: contains(matrix.runner.toolkit, 'CUDA')
run: |
RAW_TOOLKIT="${{ matrix.runner.toolkit }}"
TOOLKIT="${RAW_TOOLKIT#Nightly-}"
USE_TORCH_NIGHTLY=0
if [[ "${RAW_TOOLKIT}" == "Nightly-"* ]]; then
USE_TORCH_NIGHTLY=1
fi
if [[ ! -x "$(command -v nvcc)" ]]; then
export PATH="/usr/local/cuda/bin:${PATH}"
export LD_LIBRARY_PATH="/usr/local/cuda/lib64${LD_LIBRARY_PATH:+:${LD_LIBRARY_PATH}}"
echo "PATH=${PATH}" | tee -a "${GITHUB_ENV}"
echo "LD_LIBRARY_PATH=${LD_LIBRARY_PATH}" | tee -a "${GITHUB_ENV}"
fi
NVCC_VERSION=""
if [[ -x "$(command -v nvcc)" ]]; then
echo "\$ $(command -v nvcc) --version" && nvcc --version
NVCC_VERSION="$(nvcc --version | sed -n 's/.*release \([0-9][0-9]*\.[0-9][0-9]*\).*/\1/p' | head -n1)"
fi
if [[ "${TOOLKIT}" == "CUDA-auto" ]]; then
if [[ -z "${NVCC_VERSION}" ]]; then
echo "::error::CUDA-auto requires nvcc in PATH, but nvcc was not found or its version could not be parsed."
exit 1
fi
case "${NVCC_VERSION%%.*}" in
13) CUDA_VERSION="13.0" ;;
*)
echo "::error::CUDA-auto requires a CUDA 13 runner, but nvcc reported ${NVCC_VERSION}."
exit 1
;;
esac
else
CUDA_VERSION="${TOOLKIT##*-}"
if [[ -n "${NVCC_VERSION}" && "${NVCC_VERSION%%.*}" != "${CUDA_VERSION%%.*}" ]]; then
echo "::error::nvcc CUDA ${NVCC_VERSION} does not match requested ${TOOLKIT}."
exit 1
fi
fi
CUDA_VERSION_MAJMIN="$(echo ${CUDA_VERSION} | cut -d '.' -f-2)"
CUDA_VERSION_MAJMIN_NODOT="${CUDA_VERSION_MAJMIN//./}"
if [[ "${USE_TORCH_NIGHTLY}" == "1" ]]; then
# Use torch nightly builds
export PIP_EXTRA_INDEX_URL="https://download.pytorch.org/whl/nightly/cu${CUDA_VERSION_MAJMIN_NODOT}"
else
export PIP_EXTRA_INDEX_URL="https://download.pytorch.org/whl/cu${CUDA_VERSION_MAJMIN_NODOT}"
fi
export UV_INDEX="${PIP_EXTRA_INDEX_URL}"
echo "USE_CUDA=ON" | tee -a "${GITHUB_ENV}"
echo "CUDA_VERSION=${CUDA_VERSION}" | tee -a "${GITHUB_ENV}"
echo "CUDA_VERSION_MAJMIN=${CUDA_VERSION_MAJMIN}" | tee -a "${GITHUB_ENV}"
echo "CUDA_VERSION_MAJMIN_NODOT=${CUDA_VERSION_MAJMIN_NODOT}" | tee -a "${GITHUB_ENV}"
echo "CUDA_NVCC_VERSION=${NVCC_VERSION:-unknown}" | tee -a "${GITHUB_ENV}"
echo "PIP_EXTRA_INDEX_URL=${PIP_EXTRA_INDEX_URL}" | tee -a "${GITHUB_ENV}"
echo "UV_INDEX=${UV_INDEX}" | tee -a "${GITHUB_ENV}"
- name: Set environment (ROCm)
if: contains(matrix.runner.toolkit, 'ROCm')
run: |
TOOLKIT="${{ matrix.runner.toolkit }}"
ROCM_VERSION="${TOOLKIT##*-}"
ROCM_VERSION_MAJMIN="$(echo ${ROCM_VERSION} | cut -d '.' -f-2)"
ROCM_VERSION_MAJMIN_NODOT="${ROCM_VERSION_MAJMIN//./}"
if [[ "${TOOLKIT}" == "Nightly-"* ]]; then
# Use torch nightly builds
export PIP_EXTRA_INDEX_URL="https://download.pytorch.org/whl/nightly/rocm${ROCM_VERSION_MAJMIN}"
else
export PIP_EXTRA_INDEX_URL="https://download.pytorch.org/whl/rocm${ROCM_VERSION_MAJMIN}"
fi
export UV_INDEX="${PIP_EXTRA_INDEX_URL}"
echo "USE_ROCM=ON" | tee -a "${GITHUB_ENV}"
echo "ROCM_VERSION=${ROCM_VERSION}" | tee -a "${GITHUB_ENV}"
echo "ROCM_VERSION_MAJMIN=${ROCM_VERSION_MAJMIN}" | tee -a "${GITHUB_ENV}"
echo "ROCM_VERSION_MAJMIN_NODOT=${ROCM_VERSION_MAJMIN_NODOT}" | tee -a "${GITHUB_ENV}"
echo "PIP_EXTRA_INDEX_URL=${PIP_EXTRA_INDEX_URL}" | tee -a "${GITHUB_ENV}"
echo "UV_INDEX=${UV_INDEX}" | tee -a "${GITHUB_ENV}"
if [[ ! -x "$(command -v hipcc)" ]]; then
export PATH="/opt/rocm/bin:${PATH}"
export LD_LIBRARY_PATH="/opt/rocm/lib${LD_LIBRARY_PATH:+:${LD_LIBRARY_PATH}}"
echo "PATH=${PATH}" | tee -a "${GITHUB_ENV}"
echo "LD_LIBRARY_PATH=${LD_LIBRARY_PATH}" | tee -a "${GITHUB_ENV}"
fi
if [[ -x "$(command -v hipcc)" ]]; then
echo "\$ $(command -v hipcc) --version" && hipcc --version
else
echo "::warning::hipcc not found in PATH!"
fi
- name: Set environment (Metal)
if: contains(matrix.runner.toolkit, 'Metal')
run: |
if [[ "${{ matrix.runner.toolkit }}" == "Nightly-"* ]]; then
# Use torch nightly builds
export PIP_EXTRA_INDEX_URL="https://download.pytorch.org/whl/nightly/cpu"
export UV_INDEX="${PIP_EXTRA_INDEX_URL}"
echo "PIP_EXTRA_INDEX_URL=${PIP_EXTRA_INDEX_URL}" | tee -a "${GITHUB_ENV}"
echo "UV_INDEX=${UV_INDEX}" | tee -a "${GITHUB_ENV}"
fi
echo "USE_METAL=ON" | tee -a "${GITHUB_ENV}"
- name: Setup Python and uv with caching
id: setup-uv
uses: astral-sh/setup-uv@v7
with:
python-version: ${{ matrix.python-version }}
activate-environment: true
# Do not use cache for self-hosted runners, as it will download/upload caches which is slow.
enable-cache: ${{ !startsWith(matrix.runner.name, 'self-hosted') }}
prune-cache: ${{ !startsWith(matrix.runner.name, 'self-hosted') }}
# Use runner tool_cache for self-hosted runners
cache-local-path: ${{ env.UV_CACHE_DIR }}
ignore-nothing-to-cache: true
# Extra cache key to upload/download caches on GitHub-hosted runners
cache-suffix: uv-${{ runner.os }}-${{ runner.arch }}-${{ matrix.python-version }}-${{ matrix.runner.name }}-${{ matrix.runner.toolkit }}
cache-dependency-glob: |
pyproject.toml
requirements*.txt
.pre-commit-config.yaml
- name: Setup venv
id: setup-venv
run: |
set -o pipefail
uv pip install --upgrade pip setuptools wheel
if [[ "${UV_INDEX}" == *"/nightly/"* ]]; then
uv pip install --prerelease=allow -v torch
fi
uv pip install -v -r requirements-test.txt
echo "import torch; print(f'torch: {torch.__version__}')" | uv run --no-project --script -
if [[ "${{ matrix.runner.toolkit }}" == *"CUDA"* ]]; then
uv pip install --no-build-isolation-package=flash-attn -v -r requirements-test-cuda.txt
echo "import flash_attn; print(f'flash_attn: {flash_attn.__version__}')" | uv run --no-project --script -
elif [[ "${{ matrix.runner.toolkit }}" == *"ROCm"* ]]; then
uv pip install -v -r requirements-test-rocm.txt
elif [[ "${{ matrix.runner.toolkit }}" == *"Metal"* ]]; then
uv pip install -v -r requirements-test-metal.txt
else
echo "::error::Unknown toolkit: ${{ matrix.runner.toolkit }}"
exit 1
fi
echo "::group::torch.utils.collect_env"
uv run --no-project -m -- torch.utils.collect_env
echo "::endgroup::"
- name: Clear uv cache for self-hosted runners (if setup failed)
if: >-
${{
failure() &&
startsWith(matrix.runner.name, 'self-hosted') &&
(steps.setup-uv.conclusion == 'failure' || steps.setup-venv.conclusion == 'failure')
}}
run: |
echo "Clearing uv cache at ${UV_CACHE_DIR} due to failure."
uv cache clean
- name: Enable core dump generation (Linux / GitHub-hosted runners)
if: ${{ runner.os == 'Linux' && !startsWith(matrix.runner.name, 'self-hosted') }}
run: |
sudo sysctl -w kernel.core_pattern="core.${{ matrix.python-version }}.${{ matrix.runner.toolkit }}.%P"
sudo sysctl -w kernel.core_uses_pid=0
sudo sysctl -w fs.suid_dumpable=1
sysctl kernel.core_pattern kernel.core_uses_pid fs.suid_dumpable
- name: Enable core dump generation (macOS / GitHub-hosted runners)
if: ${{ runner.os == 'macOS' && !startsWith(matrix.runner.name, 'self-hosted') }}
run: |
sudo sysctl -w kern.corefile="core.${{ matrix.python-version }}.${{ matrix.runner.toolkit }}.%P"
sudo sysctl -w kern.coredump=1
sudo sysctl -w kern.sugid_coredump=1
sysctl kern.corefile kern.coredump kern.sugid_coredump
- name: Install project (wheel form)
run: |
uv pip install -v .
- name: Show ccache stats
if: always()
run: |
if command -v ccache >/dev/null 2>&1; then
ccache -s || true
fi
- name: Clean up stale /tmp files (self-hosted runners)
if: startsWith(matrix.runner.name, 'self-hosted')
run: |
rm -f /tmp/tmp*.so /tmp/tmp*.cu /tmp/tmp*.cubin /tmp/tmp*.cpp
rm -rf /tmp/tvm-debug-mode-tempdirs /tmp/tilelang_cutedsl_*
- name: Run examples with Python ${{ matrix.python-version }} (${{ matrix.runner.toolkit }})
if: contains(matrix.runner.toolkit, 'CUDA')
run: |
cd testing
PYTEST=(
uv run --no-project -m --
pytest --verbose --color=yes --durations=0 --showlocals --cache-clear
)
"${PYTEST[@]}" --maxfail=3 --numprocesses=8 \
--ignore=../examples/grouped_gemm/test_example_grouped_gemm.py \
--ignore=../examples/ascend \
../examples
# NVIDIA CUDA tests
- name: Run CUDA tests with Python ${{ matrix.python-version }} (${{ matrix.runner.toolkit }})
id: cuda-tests
if: contains(matrix.runner.toolkit, 'CUDA')
run: |
cd testing
PYTEST=(
uv run --no-project -m --
pytest --verbose --color=yes --durations=0 --showlocals --cache-clear
)
"${PYTEST[@]}" --maxfail=3 --numprocesses=8 \
./python
# AMD ROCm tests
- name: Run ROCm tests with Python ${{ matrix.python-version }} (${{ matrix.runner.toolkit }})
id: rocm-tests
if: contains(matrix.runner.toolkit, 'ROCm')
run: |
cd testing
PYTEST=(
uv run --no-project -m --
pytest --verbose --color=yes --durations=0 --showlocals --cache-clear
)
"${PYTEST[@]}" --maxfail=3 --numprocesses=8 \
./python
- name: Run portable examples with Python ${{ matrix.python-version }} (${{ matrix.runner.toolkit }})
if: contains(matrix.runner.toolkit, 'ROCm')
run: |
cd testing
PYTEST=(
uv run --no-project -m --
pytest --verbose --color=yes --durations=0 --showlocals --cache-clear
)
"${PYTEST[@]}" --maxfail=1 \
../examples/seer_attention/test_block_sparse_attn_tilelang.py::test_block_sparse_attn_tilelang_rocm \
../examples/topk/test_topk_tilelang.py::test_topk_tilelang_rocm \
../examples/deepseek_v32/test_tilelang_example_deepseek_v32.py::test_example_sparse_mla_fwd_rocm \
../examples/grouped_gemm/test_example_grouped_gemm.py::test_example_grouped_gemm_rocm
# Apple Metal tests
- name: Run Metal tests with Python ${{ matrix.python-version }} (${{ matrix.runner.toolkit }})
id: metal-tests
if: contains(matrix.runner.toolkit, 'Metal')
run: |
cd testing
PYTEST=(
uv run --no-project -m --
pytest --verbose --color=yes --durations=0 --showlocals --cache-clear
)
"${PYTEST[@]}" --maxfail=3 --numprocesses=8 \
-k metal \
./python
- name: List generated files
if: ${{ !cancelled() }}
run: |
find . -type f -name '*.py[co]' -delete
find . -depth -type d -name "__pycache__" -exec rm -r "{}" +
if git status --ignored --porcelain | grep -qvE '/$'; then
ls -alh $(git status --ignored --porcelain | grep -vE '/$' | grep -oE '\S+$')
fi
ascend:
name: Ascend tests
if: |
github.repository_owner == 'tile-ai' &&
(github.event_name != 'pull_request' || !github.event.pull_request.draft)
needs: [lint]
runs-on: [self-hosted, ascend]
timeout-minutes: 120
env:
PIP_USER: "0"
USE_CUDA: "OFF"
USE_ASCEND: "ON"
steps:
- name: Checkout repository
uses: actions/checkout@v7
with:
fetch-depth: 0
submodules: recursive
- name: Setup uv
uses: astral-sh/setup-uv@v7
with:
activate-environment: false
enable-cache: false
- name: Build and test Ascend
run: |
# Keep the runner's preinstalled Torch/Torch-NPU stack.
CI_BASE_PYTHON="$(python -c 'import sys; print(sys.executable)')"
for CANN_SET_ENV in \
"${ASCEND_HOME_PATH:-}/set_env.sh" \
"${ASCEND_TOOLKIT_HOME:-}/set_env.sh" \
/usr/local/Ascend/ascend-toolkit/set_env.sh \
/usr/local/Ascend/cann-*/set_env.sh; do
if [[ -f "${CANN_SET_ENV}" ]]; then
source "${CANN_SET_ENV}"
break
fi
done
export LD_LIBRARY_PATH="/usr/local/Ascend/lib64:/usr/local/Ascend/driver/lib64:/usr/local/Ascend/driver/lib64/common:/usr/local/Ascend/driver/lib64/driver:/usr/local/Ascend/driver_libs:${LD_LIBRARY_PATH:-}"
uv venv --clear --system-site-packages --python "${CI_BASE_PYTHON}" .venv
source .venv/bin/activate
export PIP_CONSTRAINT="${RUNNER_TEMP}/ascend-torch-constraints.txt"
python - <<'PY'
import os
from importlib.metadata import version
import torch
import torch_npu
assert torch.npu.is_available(), "Ascend CI requires an available NPU"
with open(os.environ["PIP_CONSTRAINT"], "w") as constraints:
for package in ("torch", "torch_npu"):
requirement = f"{package}=={version(package)}"
constraints.write(f"{requirement}\n")
print(requirement)
PY
python -m pip install --upgrade pip setuptools wheel
python -m pip install -v -r requirements-test.txt pybind11==2.12.0
python -m pip install -v .
cd testing
python -m pytest --verbose --durations=0 --maxfail=3 -n 4 ../examples/ascend
python -m pytest --verbose --durations=0 --maxfail=3 -n 32 ascend
cutedsl:
name: CuTeDSL Examples for Python 3.12 with CUDA-auto (on self-hosted-nvidia)
if: |
github.repository_owner == 'tile-ai' &&
(github.event_name != 'pull_request' || !github.event.pull_request.draft)
needs: [tests]
runs-on: [self-hosted, nvidia]
timeout-minutes: 120
steps:
- name: Checkout repository
uses: actions/checkout@v7
with:
fetch-depth: 0
submodules: recursive
- name: Set environment (self-hosted runners)
run: |
# Hide sensitive data in logs for self-hosted runners
if [[ -n "${{ secrets.SECRET_PATH_PREFIXES }}" ]]; then
echo "::add-mask::${{ secrets.SECRET_PATH_PREFIXES }}"
# Colon separated list of secrets to mask
for secret in $(echo "${{ secrets.SECRET_PATH_PREFIXES }}" | tr ':' '\n'); do
echo "::add-mask::${secret}"
done
fi
# Use runner tool_cache as cache root for self-hosted runners to avoid internet connection
# issues and to share cache between jobs.
export XDG_CACHE_HOME="${{ runner.tool_cache }}/.ci-cache-${{ github.workflow }}"
echo "XDG_CACHE_HOME=${XDG_CACHE_HOME}" | tee -a "${GITHUB_ENV}"
echo "PIP_CACHE_DIR=${XDG_CACHE_HOME}/pip" | tee -a "${GITHUB_ENV}"
echo "UV_CACHE_DIR=${XDG_CACHE_HOME}/uv" | tee -a "${GITHUB_ENV}"
echo "PRE_COMMIT_HOME=${XDG_CACHE_HOME}/pip/.pre-commit" | tee -a "${GITHUB_ENV}"
- name: Setup ccache (self-hosted runners)
run: |
if ! command -v ccache >/dev/null 2>&1; then
echo "::warning::ccache is not installed on this self-hosted runner; CMake will build without ccache."
exit 0
fi
export CCACHE_DIR="${XDG_CACHE_HOME}/ccache/self-hosted-nvidia/CUDA-auto"
mkdir -p "${CCACHE_DIR}"
echo "CCACHE_DIR=${CCACHE_DIR}" | tee -a "${GITHUB_ENV}"
echo "CCACHE_BASEDIR=${{ github.workspace }}" | tee -a "${GITHUB_ENV}"
echo "CCACHE_NOHASHDIR=true" | tee -a "${GITHUB_ENV}"
echo "CCACHE_COMPILERCHECK=content" | tee -a "${GITHUB_ENV}"
ccache -M 20G
ccache -z
ccache -s
- name: Set environment (CUDA)
run: |
TOOLKIT="CUDA-auto"
if [[ ! -x "$(command -v nvcc)" ]]; then
export PATH="/usr/local/cuda/bin:${PATH}"
export LD_LIBRARY_PATH="/usr/local/cuda/lib64${LD_LIBRARY_PATH:+:${LD_LIBRARY_PATH}}"
echo "PATH=${PATH}" | tee -a "${GITHUB_ENV}"
echo "LD_LIBRARY_PATH=${LD_LIBRARY_PATH}" | tee -a "${GITHUB_ENV}"
fi
if [[ ! -x "$(command -v nvcc)" ]]; then
echo "::error::CUDA-auto requires nvcc in PATH."
exit 1
fi
echo "\$ $(command -v nvcc) --version" && nvcc --version
NVCC_VERSION="$(nvcc --version | sed -n 's/.*release \([0-9][0-9]*\.[0-9][0-9]*\).*/\1/p' | head -n1)"
case "${NVCC_VERSION%%.*}" in
13) CUDA_VERSION="13.0" ;;
*)
echo "::error::CUDA-auto requires a CUDA 13 runner, but nvcc reported ${NVCC_VERSION}."
exit 1
;;
esac
CUDA_VERSION_MAJMIN="$(echo ${CUDA_VERSION} | cut -d '.' -f-2)"
CUDA_VERSION_MAJMIN_NODOT="${CUDA_VERSION_MAJMIN//./}"
export PIP_EXTRA_INDEX_URL="https://download.pytorch.org/whl/cu${CUDA_VERSION_MAJMIN_NODOT}"
export UV_INDEX="${PIP_EXTRA_INDEX_URL}"
echo "USE_CUDA=ON" | tee -a "${GITHUB_ENV}"
echo "CUDA_VERSION=${CUDA_VERSION}" | tee -a "${GITHUB_ENV}"
echo "CUDA_VERSION_MAJMIN=${CUDA_VERSION_MAJMIN}" | tee -a "${GITHUB_ENV}"
echo "CUDA_VERSION_MAJMIN_NODOT=${CUDA_VERSION_MAJMIN_NODOT}" | tee -a "${GITHUB_ENV}"
echo "CUDA_NVCC_VERSION=${NVCC_VERSION}" | tee -a "${GITHUB_ENV}"
echo "PIP_EXTRA_INDEX_URL=${PIP_EXTRA_INDEX_URL}" | tee -a "${GITHUB_ENV}"
echo "UV_INDEX=${UV_INDEX}" | tee -a "${GITHUB_ENV}"
- name: Setup Python and uv with caching
id: setup-uv
uses: astral-sh/setup-uv@v7
with:
python-version: "3.12"
activate-environment: true
enable-cache: false
prune-cache: false
cache-local-path: ${{ env.UV_CACHE_DIR }}
ignore-nothing-to-cache: true
cache-suffix: uv-${{ runner.os }}-${{ runner.arch }}-3.12-self-hosted-nvidia-CUDA-auto
cache-dependency-glob: |
pyproject.toml
requirements*.txt
.pre-commit-config.yaml
- name: Setup venv
id: setup-venv
run: |
set -o pipefail
uv pip install --upgrade pip setuptools wheel
uv pip install -v -r requirements-test.txt
echo "import torch; print(f'torch: {torch.__version__}')" | uv run --no-project --script -
uv pip install --no-build-isolation-package=flash-attn -v -r requirements-test-cuda.txt
echo "import flash_attn; print(f'flash_attn: {flash_attn.__version__}')" | uv run --no-project --script -
echo "::group::torch.utils.collect_env"
uv run --no-project -m -- torch.utils.collect_env
echo "::endgroup::"
- name: Clear uv cache for self-hosted runners (if setup failed)
if: >-
${{
failure() &&
(steps.setup-uv.conclusion == 'failure' || steps.setup-venv.conclusion == 'failure')
}}
run: |
echo "Clearing uv cache at ${UV_CACHE_DIR} due to failure."
uv cache clean
- name: Install project (wheel form)
run: |
uv pip install -v .
- name: Show ccache stats
if: always()
run: |
if command -v ccache >/dev/null 2>&1; then
ccache -s || true
fi
- name: Clean up stale /tmp files (self-hosted runners)
run: |
rm -f /tmp/tmp*.so /tmp/tmp*.cu /tmp/tmp*.cubin /tmp/tmp*.cpp
rm -rf /tmp/tvm-debug-mode-tempdirs /tmp/tilelang_cutedsl_*
- name: Run CuTeDSL examples with Python 3.12 (CUDA-auto)
env:
TILELANG_TARGET: cutedsl
run: |
cd testing
PYTEST=(
uv run --no-project -m --
pytest --verbose --color=yes --durations=0 --showlocals --cache-clear
)
"${PYTEST[@]}" --maxfail=3 --numprocesses=8 \
--ignore=../examples/ascend \
../examples
- name: List generated files
if: ${{ !cancelled() }}
run: |
find . -type f -name '*.py[co]' -delete
find . -depth -type d -name "__pycache__" -exec rm -r "{}" +
if git status --ignored --porcelain | grep -qvE '/$'; then
ls -alh $(git status --ignored --porcelain | grep -vE '/$' | grep -oE '\S+$')
fi