diff --git a/native/Makefile b/native/Makefile index 14256df1c..f4178b338 100644 --- a/native/Makefile +++ b/native/Makefile @@ -122,12 +122,12 @@ PYTHON_INCLUDE_FLAGS := $(shell $(PYTHON) -c "import sysconfig; paths = sysconfi # Pybind11 ABI flags PYBIND_FLAGS := $(shell $(PYTHON) -c "import pybind11; print(' '.join([f'-DPYBIND11_COMPILER_TYPE=\\\"{pybind11.get_compiler_type()}\\\"', f'-DPYBIND11_STDLIB=\\\"{pybind11.get_stdlib()}\\\"', f'-DPYBIND11_BUILD_ABI=\\\"{pybind11.get_build_abi()}\\\"']))" 2>/dev/null || echo "") -CXXFLAGS := -std=c++17 -O3 -fPIC -Wall -Wextra -Wno-unused-parameter -Wno-unused-variable \ +CXXFLAGS := -std=c++20 -O3 -fPIC -Wall -Wextra -Wno-unused-parameter -Wno-unused-variable \ -DTORCH_API_INCLUDE_EXTENSION_H -DTORCH_EXTENSION_NAME=$(TARGET) \ $(ABI_FLAG) $(PYBIND_FLAGS) \ $(TORCH_INCLUDE_FLAGS) $(PYTHON_INCLUDE_FLAGS) $(CUDA_INCLUDE_FLAGS) -Icsrc \ -MMD -MP -HOST_CXXFLAGS := -std=c++17 -O3 -fPIC -Wall -Wextra -Wno-unused-parameter -Wno-unused-variable \ +HOST_CXXFLAGS := -std=c++20 -O3 -fPIC -Wall -Wextra -Wno-unused-parameter -Wno-unused-variable \ -DTORCH_API_INCLUDE_EXTENSION_H -DTORCH_EXTENSION_NAME=$(HOST_TARGET) \ -DDMI_HOST_ONLY=1 $(ABI_FLAG) $(PYBIND_FLAGS) \ $(TORCH_INCLUDE_FLAGS) $(PYTHON_INCLUDE_FLAGS) -Icsrc \ @@ -172,7 +172,7 @@ HOST_LDFLAGS += -L$(CLICKHOUSE_CONTRIB)/lz4/lz4 -llz4 \ SM_ARCH ?= native NVCCFLAGS := \ - -std=c++17 \ + -std=c++20 \ -arch=$(SM_ARCH) \ -O2 \ -Xcompiler -fPIC \ @@ -318,12 +318,12 @@ build/conformance_spool: csrc/store/spool.cpp csrc/store/conformance_spool.cpp c # PYTHON must point at the venv interpreter: # make build/_dmi_native_sink... PYTHON=/path/to/venv/bin/python SINK_EXT := _dmi_native_sink$(shell $(PYTHON) -c "import sysconfig; print(sysconfig.get_config_var('EXT_SUFFIX'))" 2>/dev/null) -# C++20 for this target only, because ATen demands it and its floor MOVES: -# torch 2.13 refuses below C++17, torch 2.14 below C++20, and the CI pin -# `torch>=2.8,<3` resolves to whatever is newest on the day. C++20 satisfies -# both, and our own sources here compile the same under either. Everything -# else in this Makefile stays at C++17 -- this is the one target that links -# against torch. +# C++20 because ATen demands it and its floor MOVES: torch 2.13 refuses +# below C++17, torch 2.14 below C++20, and the CI pin `torch>=2.8,<3` +# resolves to whatever is newest on the day. C++20 is the floor for every +# target that includes torch's headers -- this sink, the backend and host +# objects (CXXFLAGS, HOST_CXXFLAGS) and the ring's nvcc compiles (NVCCFLAGS). +# C++17 stays only on the torch-free drivers. build/_dmi_native_sink: csrc/sink/native_pack_sink.cpp csrc/sink/record_row.cpp csrc/sink/pack_sink.cpp csrc/sink/object_key.cpp csrc/sink/bindings_sink.cpp csrc/pack/pack_builder.cpp csrc/store/spool.cpp csrc/common/json.cpp mkdir -p $(BUILD_DIR) $(CXX) -std=c++20 -O2 -fPIC -shared -DTORCH_EXTENSION_NAME=_dmi_native_sink \ diff --git a/tests/native/ring/Makefile b/tests/native/ring/Makefile index 9dfeb1222..43d78562e 100644 --- a/tests/native/ring/Makefile +++ b/tests/native/ring/Makefile @@ -1,6 +1,9 @@ PYTHON ?= python SM_ARCH ?= native -CXX_STD ?= c++17 +# C++20: test_ring_engine, test_record_consumer and +# test_clickhouse_record_sink include torch's headers, and ATen refuses +# anything older (torch 2.14's `#error C++20 or later`). +CXX_STD ?= c++20 BUILD := build INCLUDES := -I../../.. -I../../../native/csrc diff --git a/tests/test_cpu_native_build.py b/tests/test_cpu_native_build.py index e0dfcc01a..c176bbbce 100644 --- a/tests/test_cpu_native_build.py +++ b/tests/test_cpu_native_build.py @@ -1,5 +1,6 @@ from __future__ import annotations +import os from pathlib import Path import subprocess import sys @@ -29,6 +30,81 @@ def test_host_build_plan_has_no_cuda_toolchain_or_libraries(): assert forbidden not in output +def _torch_include_flags() -> set[str]: + """The ``-I`` flags that put torch's headers on a compile's search path.""" + from torch.utils.cpp_extension import include_paths + + flags = set() + for path in include_paths(): + flags.add(f"-I{path}") + flags.add(f"-I{os.path.realpath(path)}") + return flags + + +def _torch_compile_lines(makefile_dir: str, target: str) -> list[str]: + """The compiler commands a dry run of ``target`` would execute that pull + in torch's headers. + + A command is selected by the torch include path on its command line, not + by a macro only some recipes define: nvcc's flags never set + TORCH_EXTENSION_NAME, so keying on it left every .cu compile unchecked. + The flag is matched as a whole token, so the CUDA resolver's stamp + record, which carries the same directories inside one quoted string, is + not mistaken for a compile. + """ + root = Path(__file__).resolve().parents[1] + result = subprocess.run( + # PYTHON is the interpreter running this test: its torch is the one + # the build would target, and the Makefile's default `python` need + # not exist. + ["make", "-C", makefile_dir, "-B", "-n", target, f"PYTHON={sys.executable}"], + cwd=root, capture_output=True, text=True, check=False, + ) + if result.returncode != 0: + pytest.skip(f"cannot plan `{target}` here: {result.stderr[-300:]}") + torch_flags = _torch_include_flags() + return [line for line in result.stdout.splitlines() + if torch_flags.intersection(line.split())] + + +@pytest.mark.parametrize(("makefile_dir", "target"), [ + # host plans on any machine, so CI's cpu job checks it. + pytest.param("native", "host", marks=pytest.mark.cpu, id="host"), + # The full backend needs the CUDA toolchain even to be PLANNED, which a + # cpu runner does not have. It is a gpu test, not a cpu test that skips: + # the cpu gate rightly fails any skip that is not absent hardware. + pytest.param("native", "all", marks=pytest.mark.gpu, id="all"), + # The ring test binaries that link torch compile against the same + # headers, from their own Makefile and its own CXX_STD default. They + # need the CUDA resolver to be planned, so they are gpu as well. + pytest.param("tests/native/ring", "all", marks=pytest.mark.gpu, + id="ring-tests"), +]) +def test_every_torch_including_compile_requests_cxx20(makefile_dir, target): + """PyTorch's headers refuse anything older than C++20. + + ATen.h opens with ``#error C++20 or later compatible compiler is required``, + and the pinned range here (torch>=2.8,<3) resolves to a release that + enforces it. The backend and host targets were still compiled with + -std=c++17, so a fresh install could not build either one -- and CI never + noticed, because it only dry-runs `host` and the torch-free drivers never + reach ATen. This pins the flag on the plan actually executed rather than on + the Makefile's text, so it holds however the flags are assembled, and it + covers g++ and nvcc compiles alike. + + Only ``host`` is marked cpu. Planning ``all`` resolves the CUDA toolkit and + fails without one, so it runs under the gpu marker instead. + """ + lines = _torch_compile_lines(makefile_dir, target) + assert lines, f"no torch-including compile found in the `{target}` plan" + stale = [line for line in lines + if not any(f"-std={std}" in line + for std in ("c++20", "c++23", "c++26", "gnu++20", "gnu++23"))] + assert not stale, ( + f"{len(stale)} of {len(lines)} torch-including compile(s) below C++20 " + f"in `{makefile_dir}` `{target}`:\n" + stale[0][:300]) + + @pytest.mark.cpu def test_host_export_falls_back_to_cpu_backend(monkeypatch): from dmi.transport import native