From 5b0aae337b3b4b88ca174b58fc208b24a3fb8584 Mon Sep 17 00:00:00 2001 From: Alan Liu Date: Tue, 22 Sep 2026 20:13:09 -0400 Subject: [PATCH 1/4] Build the torch-including native targets as C++20 PyTorch's headers refuse anything older: ATen.h opens with `#error C++20 or later compatible compiler is required to use ATen`, and the range CI installs (torch>=2.8,<3) resolves to a release that enforces it. The full backend (CXXFLAGS), the CPU host backend (HOST_CXXFLAGS) and the CUDA ring (NVCCFLAGS) were all still -std=c++17, so on a fresh install neither `make -C native` nor `make -C native host` could compile. _dmi_native_sink was already C++20, which is why cpu-goals kept building. CI never noticed. It only DRY-RUNS `host` (test_cpu_native_build), its real builds are cpu-goals and the conformance drivers, and none of those reach ATen at C++17. The guard pins the rule on the plan make would EXECUTE rather than on the Makefile's text: every compile carrying TORCH_EXTENSION_NAME must request C++20 or later, for `host` and for `all`. It failed on the old flags (3 stale compiles in host, 8 in all) and passes now. `all` needs a CUDA toolkit to plan at all and skips where there is none, so `host` is the case CI checks. The dry runs also pass PYTHON=sys.executable: that interpreter's torch is the one the build targets, and the Makefile's default `python` need not exist. Verified from a clean native/build with torch 2.14 / CUDA 13 / RTX 4090 (sm_89): host, all and cpu-goals build with 0 errors. The remaining warnings are not new C++20 behaviour: 52 are inside torch's own headers (its bundled pybind11.h -Wattributes, and a BFloat16.h nvcc constexpr note), and the 4 in DMI code are false positives -- a signed loop index compared against py::len(), and GCC's std::optional -Wmaybe-uninitialized on a value only read under `if (max_linger_ns_ && oldest)`. cpu suite 2184 passed; the GPU ring -> NativePackSink suite 6 passed against the rebuilt backend. --- native/Makefile | 6 ++--- tests/test_cpu_native_build.py | 40 ++++++++++++++++++++++++++++++++++ 2 files changed, 43 insertions(+), 3 deletions(-) diff --git a/native/Makefile b/native/Makefile index 14256df1c..045835b0e 100644 --- a/native/Makefile +++ b/native/Makefile @@ -122,12 +122,12 @@ PYTHON_INCLUDE_FLAGS := $(shell $(PYTHON) -c "import sysconfig; paths = sysconfi # Pybind11 ABI flags PYBIND_FLAGS := $(shell $(PYTHON) -c "import pybind11; print(' '.join([f'-DPYBIND11_COMPILER_TYPE=\\\"{pybind11.get_compiler_type()}\\\"', f'-DPYBIND11_STDLIB=\\\"{pybind11.get_stdlib()}\\\"', f'-DPYBIND11_BUILD_ABI=\\\"{pybind11.get_build_abi()}\\\"']))" 2>/dev/null || echo "") -CXXFLAGS := -std=c++17 -O3 -fPIC -Wall -Wextra -Wno-unused-parameter -Wno-unused-variable \ +CXXFLAGS := -std=c++20 -O3 -fPIC -Wall -Wextra -Wno-unused-parameter -Wno-unused-variable \ -DTORCH_API_INCLUDE_EXTENSION_H -DTORCH_EXTENSION_NAME=$(TARGET) \ $(ABI_FLAG) $(PYBIND_FLAGS) \ $(TORCH_INCLUDE_FLAGS) $(PYTHON_INCLUDE_FLAGS) $(CUDA_INCLUDE_FLAGS) -Icsrc \ -MMD -MP -HOST_CXXFLAGS := -std=c++17 -O3 -fPIC -Wall -Wextra -Wno-unused-parameter -Wno-unused-variable \ +HOST_CXXFLAGS := -std=c++20 -O3 -fPIC -Wall -Wextra -Wno-unused-parameter -Wno-unused-variable \ -DTORCH_API_INCLUDE_EXTENSION_H -DTORCH_EXTENSION_NAME=$(HOST_TARGET) \ -DDMI_HOST_ONLY=1 $(ABI_FLAG) $(PYBIND_FLAGS) \ $(TORCH_INCLUDE_FLAGS) $(PYTHON_INCLUDE_FLAGS) -Icsrc \ @@ -172,7 +172,7 @@ HOST_LDFLAGS += -L$(CLICKHOUSE_CONTRIB)/lz4/lz4 -llz4 \ SM_ARCH ?= native NVCCFLAGS := \ - -std=c++17 \ + -std=c++20 \ -arch=$(SM_ARCH) \ -O2 \ -Xcompiler -fPIC \ diff --git a/tests/test_cpu_native_build.py b/tests/test_cpu_native_build.py index e0dfcc01a..5674e3e9f 100644 --- a/tests/test_cpu_native_build.py +++ b/tests/test_cpu_native_build.py @@ -29,6 +29,46 @@ def test_host_build_plan_has_no_cuda_toolchain_or_libraries(): assert forbidden not in output +def _torch_compile_lines(target: str) -> list[str]: + """The compile commands a dry run of ``target`` would execute that pull in + torch's headers -- the ones carrying TORCH_EXTENSION_NAME.""" + root = Path(__file__).resolve().parents[1] + result = subprocess.run( + # PYTHON is the interpreter running this test: its torch is the one + # the build would target, and the Makefile's default `python` need + # not exist. + ["make", "-C", "native", "-B", "-n", target, f"PYTHON={sys.executable}"], + cwd=root, capture_output=True, text=True, check=False, + ) + if result.returncode != 0: + pytest.skip(f"cannot plan `{target}` here: {result.stderr[-300:]}") + return [line for line in result.stdout.splitlines() + if "-DTORCH_EXTENSION_NAME=" in line and " -c " in f" {line} "] + + +@pytest.mark.cpu +@pytest.mark.parametrize("target", ["host", "all"]) +def test_every_torch_including_compile_requests_cxx20(target): + """PyTorch's headers refuse anything older than C++20. + + ATen.h opens with ``#error C++20 or later compatible compiler is required``, + and the pinned range here (torch>=2.8,<3) resolves to a release that + enforces it. The backend and host targets were still compiled with + -std=c++17, so a fresh install could not build either one -- and CI never + noticed, because it only dry-runs `host` and the torch-free drivers never + reach ATen. This pins the flag on the plan actually executed rather than on + the Makefile's text, so it holds however the flags are assembled. + """ + lines = _torch_compile_lines(target) + assert lines, f"no torch-including compile found in the `{target}` plan" + stale = [line for line in lines + if not any(f"-std={std}" in line + for std in ("c++20", "c++23", "c++26", "gnu++20", "gnu++23"))] + assert not stale, ( + f"{len(stale)} torch-including compile(s) below C++20 in `{target}`:\n" + + stale[0][:300]) + + @pytest.mark.cpu def test_host_export_falls_back_to_cpu_backend(monkeypatch): from dmi.transport import native From 3d2120731657ba5b087c52fe9c08c84b4b9d6ee2 Mon Sep 17 00:00:00 2001 From: Alan Liu Date: Tue, 22 Sep 2026 20:29:12 -0400 Subject: [PATCH 2/4] Mark the full-backend C++20 guard gpu, not cpu The cpu gate fails any skip whose reason is not absent hardware, and it was right to fail this one. Planning `all` resolves the CUDA toolkit and stops without it, so on a cpu runner the case skipped with "CUDA toolkit resolution failed" -- a missing build prerequisite, which is exactly the kind of skip the gate exists to refuse. Rewording the reason to match its allow-list would have hidden it rather than fixed it. The case was never a cpu test. It now carries the gpu marker, so the cpu job does not select it, while host -- which plans anywhere -- stays cpu and is what CI checks. With C++17 restored on HOST_CXXFLAGS the cpu case still fails (3 stale compiles), so the move did not weaken the guard. Local run through the same gate: 2183 collected, 0 unexplained skips. --- tests/test_cpu_native_build.py | 13 +++++++++++-- 1 file changed, 11 insertions(+), 2 deletions(-) diff --git a/tests/test_cpu_native_build.py b/tests/test_cpu_native_build.py index 5674e3e9f..1a3d505f9 100644 --- a/tests/test_cpu_native_build.py +++ b/tests/test_cpu_native_build.py @@ -46,8 +46,14 @@ def _torch_compile_lines(target: str) -> list[str]: if "-DTORCH_EXTENSION_NAME=" in line and " -c " in f" {line} "] -@pytest.mark.cpu -@pytest.mark.parametrize("target", ["host", "all"]) +@pytest.mark.parametrize("target", [ + # host plans on any machine, so CI's cpu job checks it. + pytest.param("host", marks=pytest.mark.cpu), + # The full backend needs the CUDA toolchain even to be PLANNED, which a + # cpu runner does not have. It is a gpu test, not a cpu test that skips: + # the cpu gate rightly fails any skip that is not absent hardware. + pytest.param("all", marks=pytest.mark.gpu), +]) def test_every_torch_including_compile_requests_cxx20(target): """PyTorch's headers refuse anything older than C++20. @@ -58,6 +64,9 @@ def test_every_torch_including_compile_requests_cxx20(target): noticed, because it only dry-runs `host` and the torch-free drivers never reach ATen. This pins the flag on the plan actually executed rather than on the Makefile's text, so it holds however the flags are assembled. + + Only ``host`` is marked cpu. Planning ``all`` resolves the CUDA toolkit and + fails without one, so it runs under the gpu marker instead. """ lines = _torch_compile_lines(target) assert lines, f"no torch-including compile found in the `{target}` plan" From d10d3daf842bd6d90fa8c71e680c3f1213fc8f67 Mon Sep 17 00:00:00 2001 From: Alan Liu Date: Wed, 23 Sep 2026 19:07:19 -0400 Subject: [PATCH 3/4] Check the nvcc compiles in the C++20 guard The guard picked its compile lines by -DTORCH_EXTENSION_NAME=, which only CXXFLAGS and HOST_CXXFLAGS define. NVCCFLAGS never sets it, so the three ring .cu compiles in the `all` plan -- which include torch's headers just the same -- were never checked. Reverting only NVCCFLAGS to -std=c++17 left the guard green (2 passed) while a real `make build/ring/ring_engine_py.o` failed with ATen's `#error C++20 or later compatible compiler is required`. The guard now selects every command whose tokens carry a torch include path (torch.utils.cpp_extension.include_paths()), so g++ and nvcc compiles are checked alike: the `all` plan goes from 8 checked compiles to 11. Evidence, on scratch copies of the tree: - PR head: 2 passed (host, all). - NVCCFLAGS alone at c++17: old guard 2 passed; new guard fails `all` with "3 of 11 torch-including compile(s) below C++20". - origin/main: fails host (3 of 3) and all (11 of 11). Also reword the stale comment above the sink target: C++20 is the floor for every torch-including target, and C++17 stays only on the torch-free drivers. Claude-Session: https://claude.ai/code/session_01PcY9QS1FAehHTzkjdpvN6Y --- native/Makefile | 12 ++++----- tests/test_cpu_native_build.py | 48 +++++++++++++++++++++++++--------- 2 files changed, 41 insertions(+), 19 deletions(-) diff --git a/native/Makefile b/native/Makefile index 045835b0e..f4178b338 100644 --- a/native/Makefile +++ b/native/Makefile @@ -318,12 +318,12 @@ build/conformance_spool: csrc/store/spool.cpp csrc/store/conformance_spool.cpp c # PYTHON must point at the venv interpreter: # make build/_dmi_native_sink... PYTHON=/path/to/venv/bin/python SINK_EXT := _dmi_native_sink$(shell $(PYTHON) -c "import sysconfig; print(sysconfig.get_config_var('EXT_SUFFIX'))" 2>/dev/null) -# C++20 for this target only, because ATen demands it and its floor MOVES: -# torch 2.13 refuses below C++17, torch 2.14 below C++20, and the CI pin -# `torch>=2.8,<3` resolves to whatever is newest on the day. C++20 satisfies -# both, and our own sources here compile the same under either. Everything -# else in this Makefile stays at C++17 -- this is the one target that links -# against torch. +# C++20 because ATen demands it and its floor MOVES: torch 2.13 refuses +# below C++17, torch 2.14 below C++20, and the CI pin `torch>=2.8,<3` +# resolves to whatever is newest on the day. C++20 is the floor for every +# target that includes torch's headers -- this sink, the backend and host +# objects (CXXFLAGS, HOST_CXXFLAGS) and the ring's nvcc compiles (NVCCFLAGS). +# C++17 stays only on the torch-free drivers. build/_dmi_native_sink: csrc/sink/native_pack_sink.cpp csrc/sink/record_row.cpp csrc/sink/pack_sink.cpp csrc/sink/object_key.cpp csrc/sink/bindings_sink.cpp csrc/pack/pack_builder.cpp csrc/store/spool.cpp csrc/common/json.cpp mkdir -p $(BUILD_DIR) $(CXX) -std=c++20 -O2 -fPIC -shared -DTORCH_EXTENSION_NAME=_dmi_native_sink \ diff --git a/tests/test_cpu_native_build.py b/tests/test_cpu_native_build.py index 1a3d505f9..26759cffd 100644 --- a/tests/test_cpu_native_build.py +++ b/tests/test_cpu_native_build.py @@ -1,5 +1,6 @@ from __future__ import annotations +import os from pathlib import Path import subprocess import sys @@ -29,32 +30,52 @@ def test_host_build_plan_has_no_cuda_toolchain_or_libraries(): assert forbidden not in output -def _torch_compile_lines(target: str) -> list[str]: - """The compile commands a dry run of ``target`` would execute that pull in - torch's headers -- the ones carrying TORCH_EXTENSION_NAME.""" +def _torch_include_flags() -> set[str]: + """The ``-I`` flags that put torch's headers on a compile's search path.""" + from torch.utils.cpp_extension import include_paths + + flags = set() + for path in include_paths(): + flags.add(f"-I{path}") + flags.add(f"-I{os.path.realpath(path)}") + return flags + + +def _torch_compile_lines(makefile_dir: str, target: str) -> list[str]: + """The compiler commands a dry run of ``target`` would execute that pull + in torch's headers. + + A command is selected by the torch include path on its command line, not + by a macro only some recipes define: nvcc's flags never set + TORCH_EXTENSION_NAME, so keying on it left every .cu compile unchecked. + The flag is matched as a whole token, so the CUDA resolver's stamp + record, which carries the same directories inside one quoted string, is + not mistaken for a compile. + """ root = Path(__file__).resolve().parents[1] result = subprocess.run( # PYTHON is the interpreter running this test: its torch is the one # the build would target, and the Makefile's default `python` need # not exist. - ["make", "-C", "native", "-B", "-n", target, f"PYTHON={sys.executable}"], + ["make", "-C", makefile_dir, "-B", "-n", target, f"PYTHON={sys.executable}"], cwd=root, capture_output=True, text=True, check=False, ) if result.returncode != 0: pytest.skip(f"cannot plan `{target}` here: {result.stderr[-300:]}") + torch_flags = _torch_include_flags() return [line for line in result.stdout.splitlines() - if "-DTORCH_EXTENSION_NAME=" in line and " -c " in f" {line} "] + if torch_flags.intersection(line.split())] -@pytest.mark.parametrize("target", [ +@pytest.mark.parametrize(("makefile_dir", "target"), [ # host plans on any machine, so CI's cpu job checks it. - pytest.param("host", marks=pytest.mark.cpu), + pytest.param("native", "host", marks=pytest.mark.cpu, id="host"), # The full backend needs the CUDA toolchain even to be PLANNED, which a # cpu runner does not have. It is a gpu test, not a cpu test that skips: # the cpu gate rightly fails any skip that is not absent hardware. - pytest.param("all", marks=pytest.mark.gpu), + pytest.param("native", "all", marks=pytest.mark.gpu, id="all"), ]) -def test_every_torch_including_compile_requests_cxx20(target): +def test_every_torch_including_compile_requests_cxx20(makefile_dir, target): """PyTorch's headers refuse anything older than C++20. ATen.h opens with ``#error C++20 or later compatible compiler is required``, @@ -63,19 +84,20 @@ def test_every_torch_including_compile_requests_cxx20(target): -std=c++17, so a fresh install could not build either one -- and CI never noticed, because it only dry-runs `host` and the torch-free drivers never reach ATen. This pins the flag on the plan actually executed rather than on - the Makefile's text, so it holds however the flags are assembled. + the Makefile's text, so it holds however the flags are assembled, and it + covers g++ and nvcc compiles alike. Only ``host`` is marked cpu. Planning ``all`` resolves the CUDA toolkit and fails without one, so it runs under the gpu marker instead. """ - lines = _torch_compile_lines(target) + lines = _torch_compile_lines(makefile_dir, target) assert lines, f"no torch-including compile found in the `{target}` plan" stale = [line for line in lines if not any(f"-std={std}" in line for std in ("c++20", "c++23", "c++26", "gnu++20", "gnu++23"))] assert not stale, ( - f"{len(stale)} torch-including compile(s) below C++20 in `{target}`:\n" - + stale[0][:300]) + f"{len(stale)} of {len(lines)} torch-including compile(s) below C++20 " + f"in `{makefile_dir}` `{target}`:\n" + stale[0][:300]) @pytest.mark.cpu From 460e84dc603ddaadb8603dc305885ae9a2326e52 Mon Sep 17 00:00:00 2001 From: Alan Liu Date: Wed, 23 Sep 2026 19:07:52 -0400 Subject: [PATCH 4/4] Build the ring test binaries as C++20 tests/native/ring/Makefile defaulted CXX_STD to c++17, and three of its binaries -- test_ring_engine, test_record_consumer and test_clickhouse_record_sink -- include torch's headers, so against torch 2.14 each one stops at ATen's `#error C++20 or later compatible compiler is required`. The default is now c++20; CXX_STD can still be overridden. The C++20 guard now also dry-runs this Makefile's `all` (gpu-marked, as planning it runs the CUDA resolver), so the default cannot slip back. Evidence: - Guard with the ring case, Makefile still c++17: 1 failed, 12 passed ("3 of 3 torch-including compile(s) below C++20 in tests/native/ring"). With c++20: 13 passed. - `make -C tests/native/ring build/test_record_consumer SM_ARCH=sm_89` on a scratch copy of the tree: rc=2 with the ATen #error at c++17, rc=0 at c++20. `make all` at c++20 builds all six ring binaries. Claude-Session: https://claude.ai/code/session_01PcY9QS1FAehHTzkjdpvN6Y --- tests/native/ring/Makefile | 5 ++++- tests/test_cpu_native_build.py | 5 +++++ 2 files changed, 9 insertions(+), 1 deletion(-) diff --git a/tests/native/ring/Makefile b/tests/native/ring/Makefile index 9dfeb1222..43d78562e 100644 --- a/tests/native/ring/Makefile +++ b/tests/native/ring/Makefile @@ -1,6 +1,9 @@ PYTHON ?= python SM_ARCH ?= native -CXX_STD ?= c++17 +# C++20: test_ring_engine, test_record_consumer and +# test_clickhouse_record_sink include torch's headers, and ATen refuses +# anything older (torch 2.14's `#error C++20 or later`). +CXX_STD ?= c++20 BUILD := build INCLUDES := -I../../.. -I../../../native/csrc diff --git a/tests/test_cpu_native_build.py b/tests/test_cpu_native_build.py index 26759cffd..c176bbbce 100644 --- a/tests/test_cpu_native_build.py +++ b/tests/test_cpu_native_build.py @@ -74,6 +74,11 @@ def _torch_compile_lines(makefile_dir: str, target: str) -> list[str]: # cpu runner does not have. It is a gpu test, not a cpu test that skips: # the cpu gate rightly fails any skip that is not absent hardware. pytest.param("native", "all", marks=pytest.mark.gpu, id="all"), + # The ring test binaries that link torch compile against the same + # headers, from their own Makefile and its own CXX_STD default. They + # need the CUDA resolver to be planned, so they are gpu as well. + pytest.param("tests/native/ring", "all", marks=pytest.mark.gpu, + id="ring-tests"), ]) def test_every_torch_including_compile_requests_cxx20(makefile_dir, target): """PyTorch's headers refuse anything older than C++20.