diff --git a/cuda_core/cuda/core/_linker.pyi b/cuda_core/cuda/core/_linker.pyi index 9ff5dfd53c7..c444e82a59b 100644 --- a/cuda_core/cuda/core/_linker.pyi +++ b/cuda_core/cuda/core/_linker.pyi @@ -153,7 +153,9 @@ class LinkerOptions: incremental : bool, optional Perform an incremental link. The result can be passed directly to a later :class:`Linker`. Requires nvJitLink 13.2 or newer - and is not supported by the driver linker backend. + and is not supported by the driver linker backend. Intermediate + caching is disabled automatically when combined with link-time + optimization. Default: False. ptx : bool, optional Emit PTX after linking instead of CUBIN; only supported with ``link_time_optimization=True``. @@ -198,6 +200,8 @@ class LinkerOptions: Default: 1. no_cache : bool, optional Do not cache the intermediate steps of nvJitLink. + This is also enabled automatically for incremental links that use + link-time optimization. Default: False. numba_debug : bool, optional Non-functional. ``numba_debug`` is an NVVM/NVRTC *compiler* option; diff --git a/cuda_core/cuda/core/_linker.pyx b/cuda_core/cuda/core/_linker.pyx index e5deeb07b10..139d1082bf1 100644 --- a/cuda_core/cuda/core/_linker.pyx +++ b/cuda_core/cuda/core/_linker.pyx @@ -271,7 +271,9 @@ class LinkerOptions: incremental : bool, optional Perform an incremental link. The result can be passed directly to a later :class:`Linker`. Requires nvJitLink 13.2 or newer - and is not supported by the driver linker backend. + and is not supported by the driver linker backend. Intermediate + caching is disabled automatically when combined with link-time + optimization. Default: False. ptx : bool, optional Emit PTX after linking instead of CUBIN; only supported with ``link_time_optimization=True``. @@ -316,6 +318,8 @@ class LinkerOptions: Default: 1. no_cache : bool, optional Do not cache the intermediate steps of nvJitLink. + This is also enabled automatically for incremental links that use + link-time optimization. Default: False. numba_debug : bool, optional Non-functional. ``numba_debug`` is an NVVM/NVRTC *compiler* option; @@ -431,7 +435,10 @@ class LinkerOptions: options.append(f"-split-compile={self.split_compile}") if self.split_compile_extended is not None: options.append(f"-split-compile-extended={self.split_compile_extended}") - if self.no_cache is True: + # Repeated incremental LTO links can crash in nvJitLink's intermediate + # cache (reproduced with 13.4.92 on Windows). Avoid reusing cached + # state for each incremental stage. + if self.no_cache is True or (self.incremental and self.link_time_optimization): options.append("-no-cache") if as_bytes: diff --git a/cuda_core/tests/test_linker.py b/cuda_core/tests/test_linker.py index e25578c3dd1..64c2630e5b5 100644 --- a/cuda_core/tests/test_linker.py +++ b/cuda_core/tests/test_linker.py @@ -218,6 +218,19 @@ def test_linker_options_incremental_as_bytes(value, expected_count): assert options.as_bytes().count(b"-r") == expected_count +@pytest.mark.agent_authored(model="gpt-6") +@pytest.mark.skipif(is_culink_backend, reason="as_bytes() only supported for nvjitlink backend") +def test_linker_options_incremental_lto_disables_cache(): + options = LinkerOptions(arch="sm_80", incremental=True, link_time_optimization=True) + assert b"-no-cache" in options.as_bytes() + + options = LinkerOptions(arch="sm_80", incremental=True) + assert b"-no-cache" not in options.as_bytes() + + options = LinkerOptions(arch="sm_80", link_time_optimization=True) + assert b"-no-cache" not in options.as_bytes() + + @pytest.mark.parametrize("backend", ("invalid", "driver")) def test_linker_options_as_bytes_invalid_backend(backend): """Test LinkerOptions.as_bytes() with invalid backend"""