Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
18 changes: 9 additions & 9 deletions native/Makefile
Original file line number Diff line number Diff line change
Expand Up @@ -122,12 +122,12 @@ PYTHON_INCLUDE_FLAGS := $(shell $(PYTHON) -c "import sysconfig; paths = sysconfi
# Pybind11 ABI flags
PYBIND_FLAGS := $(shell $(PYTHON) -c "import pybind11; print(' '.join([f'-DPYBIND11_COMPILER_TYPE=\\\"{pybind11.get_compiler_type()}\\\"', f'-DPYBIND11_STDLIB=\\\"{pybind11.get_stdlib()}\\\"', f'-DPYBIND11_BUILD_ABI=\\\"{pybind11.get_build_abi()}\\\"']))" 2>/dev/null || echo "")

CXXFLAGS := -std=c++17 -O3 -fPIC -Wall -Wextra -Wno-unused-parameter -Wno-unused-variable \
CXXFLAGS := -std=c++20 -O3 -fPIC -Wall -Wextra -Wno-unused-parameter -Wno-unused-variable \
-DTORCH_API_INCLUDE_EXTENSION_H -DTORCH_EXTENSION_NAME=$(TARGET) \
$(ABI_FLAG) $(PYBIND_FLAGS) \
$(TORCH_INCLUDE_FLAGS) $(PYTHON_INCLUDE_FLAGS) $(CUDA_INCLUDE_FLAGS) -Icsrc \
-MMD -MP
HOST_CXXFLAGS := -std=c++17 -O3 -fPIC -Wall -Wextra -Wno-unused-parameter -Wno-unused-variable \
HOST_CXXFLAGS := -std=c++20 -O3 -fPIC -Wall -Wextra -Wno-unused-parameter -Wno-unused-variable \
-DTORCH_API_INCLUDE_EXTENSION_H -DTORCH_EXTENSION_NAME=$(HOST_TARGET) \
-DDMI_HOST_ONLY=1 $(ABI_FLAG) $(PYBIND_FLAGS) \
$(TORCH_INCLUDE_FLAGS) $(PYTHON_INCLUDE_FLAGS) -Icsrc \
Expand Down Expand Up @@ -172,7 +172,7 @@ HOST_LDFLAGS += -L$(CLICKHOUSE_CONTRIB)/lz4/lz4 -llz4 \
SM_ARCH ?= native

NVCCFLAGS := \
-std=c++17 \
-std=c++20 \
-arch=$(SM_ARCH) \
-O2 \
-Xcompiler -fPIC \
Expand Down Expand Up @@ -318,12 +318,12 @@ build/conformance_spool: csrc/store/spool.cpp csrc/store/conformance_spool.cpp c
# PYTHON must point at the venv interpreter:
# make build/_dmi_native_sink... PYTHON=/path/to/venv/bin/python
SINK_EXT := _dmi_native_sink$(shell $(PYTHON) -c "import sysconfig; print(sysconfig.get_config_var('EXT_SUFFIX'))" 2>/dev/null)
# C++20 for this target only, because ATen demands it and its floor MOVES:
# torch 2.13 refuses below C++17, torch 2.14 below C++20, and the CI pin
# `torch>=2.8,<3` resolves to whatever is newest on the day. C++20 satisfies
# both, and our own sources here compile the same under either. Everything
# else in this Makefile stays at C++17 -- this is the one target that links
# against torch.
# C++20 because ATen demands it and its floor MOVES: torch 2.13 refuses
# below C++17, torch 2.14 below C++20, and the CI pin `torch>=2.8,<3`
# resolves to whatever is newest on the day. C++20 is the floor for every
# target that includes torch's headers -- this sink, the backend and host
# objects (CXXFLAGS, HOST_CXXFLAGS) and the ring's nvcc compiles (NVCCFLAGS).
# C++17 stays only on the torch-free drivers.
build/_dmi_native_sink: csrc/sink/native_pack_sink.cpp csrc/sink/record_row.cpp csrc/sink/pack_sink.cpp csrc/sink/object_key.cpp csrc/sink/bindings_sink.cpp csrc/pack/pack_builder.cpp csrc/store/spool.cpp csrc/common/json.cpp
mkdir -p $(BUILD_DIR)
$(CXX) -std=c++20 -O2 -fPIC -shared -DTORCH_EXTENSION_NAME=_dmi_native_sink \
Expand Down
5 changes: 4 additions & 1 deletion tests/native/ring/Makefile
Original file line number Diff line number Diff line change
@@ -1,6 +1,9 @@
PYTHON ?= python
SM_ARCH ?= native
CXX_STD ?= c++17
# C++20: test_ring_engine, test_record_consumer and
# test_clickhouse_record_sink include torch's headers, and ATen refuses
# anything older (torch 2.14's `#error C++20 or later`).
CXX_STD ?= c++20
BUILD := build
INCLUDES := -I../../.. -I../../../native/csrc

Expand Down
76 changes: 76 additions & 0 deletions tests/test_cpu_native_build.py
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
from __future__ import annotations

import os
from pathlib import Path
import subprocess
import sys
Expand Down Expand Up @@ -29,6 +30,81 @@ def test_host_build_plan_has_no_cuda_toolchain_or_libraries():
assert forbidden not in output


def _torch_include_flags() -> set[str]:
"""The ``-I`` flags that put torch's headers on a compile's search path."""
from torch.utils.cpp_extension import include_paths

flags = set()
for path in include_paths():
flags.add(f"-I{path}")
flags.add(f"-I{os.path.realpath(path)}")
return flags


def _torch_compile_lines(makefile_dir: str, target: str) -> list[str]:
"""The compiler commands a dry run of ``target`` would execute that pull
in torch's headers.

A command is selected by the torch include path on its command line, not
by a macro only some recipes define: nvcc's flags never set
TORCH_EXTENSION_NAME, so keying on it left every .cu compile unchecked.
The flag is matched as a whole token, so the CUDA resolver's stamp
record, which carries the same directories inside one quoted string, is
not mistaken for a compile.
"""
root = Path(__file__).resolve().parents[1]
result = subprocess.run(
# PYTHON is the interpreter running this test: its torch is the one
# the build would target, and the Makefile's default `python` need
# not exist.
["make", "-C", makefile_dir, "-B", "-n", target, f"PYTHON={sys.executable}"],
cwd=root, capture_output=True, text=True, check=False,
)
if result.returncode != 0:
pytest.skip(f"cannot plan `{target}` here: {result.stderr[-300:]}")
torch_flags = _torch_include_flags()
return [line for line in result.stdout.splitlines()
if torch_flags.intersection(line.split())]


@pytest.mark.parametrize(("makefile_dir", "target"), [
# host plans on any machine, so CI's cpu job checks it.
pytest.param("native", "host", marks=pytest.mark.cpu, id="host"),
# The full backend needs the CUDA toolchain even to be PLANNED, which a
# cpu runner does not have. It is a gpu test, not a cpu test that skips:
# the cpu gate rightly fails any skip that is not absent hardware.
pytest.param("native", "all", marks=pytest.mark.gpu, id="all"),
# The ring test binaries that link torch compile against the same
# headers, from their own Makefile and its own CXX_STD default. They
# need the CUDA resolver to be planned, so they are gpu as well.
pytest.param("tests/native/ring", "all", marks=pytest.mark.gpu,
id="ring-tests"),
])
def test_every_torch_including_compile_requests_cxx20(makefile_dir, target):
"""PyTorch's headers refuse anything older than C++20.

ATen.h opens with ``#error C++20 or later compatible compiler is required``,
and the pinned range here (torch>=2.8,<3) resolves to a release that
enforces it. The backend and host targets were still compiled with
-std=c++17, so a fresh install could not build either one -- and CI never
noticed, because it only dry-runs `host` and the torch-free drivers never
reach ATen. This pins the flag on the plan actually executed rather than on
the Makefile's text, so it holds however the flags are assembled, and it
covers g++ and nvcc compiles alike.

Only ``host`` is marked cpu. Planning ``all`` resolves the CUDA toolkit and
fails without one, so it runs under the gpu marker instead.
"""
lines = _torch_compile_lines(makefile_dir, target)
assert lines, f"no torch-including compile found in the `{target}` plan"
stale = [line for line in lines
if not any(f"-std={std}" in line
for std in ("c++20", "c++23", "c++26", "gnu++20", "gnu++23"))]
assert not stale, (
f"{len(stale)} of {len(lines)} torch-including compile(s) below C++20 "
f"in `{makefile_dir}` `{target}`:\n" + stale[0][:300])


@pytest.mark.cpu
def test_host_export_falls_back_to_cpu_backend(monkeypatch):
from dmi.transport import native
Expand Down
Loading