From 0d94843b5e00b82bc9243695e1833fe542269c01 Mon Sep 17 00:00:00 2001 From: Lei Sun Date: Wed, 7 Oct 2026 00:01:20 +0000 Subject: [PATCH 1/3] Document pinned allocation tuning for fitted caches --- docs/source/icl.md | 29 +++++++++++++++++++++++++++++ 1 file changed, 29 insertions(+) diff --git a/docs/source/icl.md b/docs/source/icl.md index fbab8c151..b5325bb70 100644 --- a/docs/source/icl.md +++ b/docs/source/icl.md @@ -73,6 +73,35 @@ Use one-shot {py:meth}`~sdm.models.ICLModel.forward` calls for one-time calls wh Unlike {py:meth}`~sdm.models.ICLModel.forward`, {py:meth}`~sdm.models.ICLModel.predict` does not support gradient-based fine-tuning and raises if the model is in train mode. +### Pinned host memory for fitted caches + +For CUDA fits with multiple estimators, SDM can store fitted state in pinned host memory, which supports asynchronous transfers to the GPU. +PyTorch normally rounds individual pinned allocations up to a power of two. +For example, a 3 MiB tensor can occupy a 4 MiB allocation. + +With PyTorch 2.13 or later, set `pinned_max_round_threshold_mb:1` before starting Python to use exact allocation sizes above 1 MiB: + +```bash +alloc_conf="${PYTORCH_ALLOC_CONF:-${PYTORCH_CUDA_ALLOC_CONF:-}}" +PYTORCH_ALLOC_CONF="${alloc_conf:+${alloc_conf},}pinned_max_round_threshold_mb:1" \ + python inference.py +``` + +Replace `inference.py` with your inference script. +The command preserves other allocator options from `PYTORCH_ALLOC_CONF`, or from its legacy alias `PYTORCH_CUDA_ALLOC_CONF`. +If the existing configuration already specifies `pinned_max_round_threshold_mb`, edit that value instead of appending it again. +Older PyTorch versions can reject this option. + +The setting applies to all pinned allocations in the process. +It changes allocation capacity without changing tensor contents or model arithmetic. +Exact sizes can reduce buffer reuse when allocation sizes vary, so compare fit time, prediction time, and host memory on your workload. + +This setting does not limit how much freed pinned memory PyTorch retains for reuse. +Keep `pinned_max_cached_size_mb` unchanged when evaluating rounding alone. +Reducing that separate limit can require expensive pinned allocations during later fits. +The live cache still needs space for its full tensor payload. +See [PyTorch's pinned-memory allocator options](https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage) for details. + ## Model Concepts Structured data foundation models are not bound to a specific task type. From c6711286492fd12ecc8ab0b0628a3f8544183ae0 Mon Sep 17 00:00:00 2001 From: Lei Sun Date: Wed, 7 Oct 2026 00:36:49 +0000 Subject: [PATCH 2/3] Enable exact-size pinned allocations for fitted caches --- docs/source/icl.md | 15 +++-- sdm/_memory.py | 18 ++++++ sdm/models/base.py | 2 + test/test_memory.py | 132 ++++++++++++++++++++++++++++++++++++++++++++ 4 files changed, 161 insertions(+), 6 deletions(-) create mode 100644 test/test_memory.py diff --git a/docs/source/icl.md b/docs/source/icl.md index b5325bb70..84a3a3002 100644 --- a/docs/source/icl.md +++ b/docs/source/icl.md @@ -79,17 +79,20 @@ For CUDA fits with multiple estimators, SDM can store fitted state in pinned hos PyTorch normally rounds individual pinned allocations up to a power of two. For example, a 3 MiB tensor can occupy a 4 MiB allocation. -With PyTorch 2.13 or later, set `pinned_max_round_threshold_mb:1` before starting Python to use exact allocation sizes above 1 MiB: +With PyTorch 2.13 or later, SDM automatically sets `pinned_max_round_threshold_mb:1` before offloading a fitted cache. +Allocations above 1 MiB then use exact sizes. +SDM preserves the current allocator options, including settings applied after CUDA initialization. +An explicit rounding threshold takes precedence over SDM's default. +Older PyTorch versions keep their existing allocation behavior. + +To select a different threshold, configure PyTorch before starting Python: ```bash -alloc_conf="${PYTORCH_ALLOC_CONF:-${PYTORCH_CUDA_ALLOC_CONF:-}}" -PYTORCH_ALLOC_CONF="${alloc_conf:+${alloc_conf},}pinned_max_round_threshold_mb:1" \ - python inference.py +PYTORCH_ALLOC_CONF=pinned_max_round_threshold_mb:128 python inference.py ``` Replace `inference.py` with your inference script. -The command preserves other allocator options from `PYTORCH_ALLOC_CONF`, or from its legacy alias `PYTORCH_CUDA_ALLOC_CONF`. -If the existing configuration already specifies `pinned_max_round_threshold_mb`, edit that value instead of appending it again. +If you already configure the allocator, add or edit the threshold in that configuration while keeping your other options. Older PyTorch versions can reject this option. The setting applies to all pinned allocations in the process. diff --git a/sdm/_memory.py b/sdm/_memory.py index af8828c47..1d4198a6f 100644 --- a/sdm/_memory.py +++ b/sdm/_memory.py @@ -6,6 +6,24 @@ import torch +def configure_pinned_memory() -> None: + r"""Avoid rounding large pinned allocations unless explicitly configured. + + The allocator setting applies process-wide and retains freed blocks for + reuse. PyTorch versions before 2.13 do not support the rounding threshold. + """ + version = tuple(int(part) for part in torch.__version__.split(".")[:2]) + if version < (2, 13): + return + settings = torch._C._accelerator_getAllocatorSettings() + if "pinned_max_round_threshold_mb" in settings: + return + setting = "pinned_max_round_threshold_mb:1" + torch._C._accelerator_setAllocatorSettings( + f"{settings},{setting}" if settings else setting + ) + + def chunk_memory_limit(device: torch.device) -> int: r"""Bytes one chunk of a chunked operation may occupy on a CUDA device. diff --git a/sdm/models/base.py b/sdm/models/base.py index 83d9bc3f5..0f6a8c4b7 100644 --- a/sdm/models/base.py +++ b/sdm/models/base.py @@ -20,6 +20,7 @@ TaskLike, ) from sdm._inference import inference_mode +from sdm._memory import configure_pinned_memory from sdm._warnings import warn_once from sdm.cache import Cache from sdm.models.callback import Callback @@ -312,6 +313,7 @@ def fit( ) if x.is_cuda and len(contexts) > 1: + configure_pinned_memory() try: # Copy to pinned CPU memory: batch_cache = batch_cache._apply_tensor( lambda tensor: torch.ops.aten._to_copy.default( diff --git a/test/test_memory.py b/test/test_memory.py new file mode 100644 index 000000000..554c791d8 --- /dev/null +++ b/test/test_memory.py @@ -0,0 +1,132 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +import json +import os +import subprocess +import sys + +import pytest +import torch + +from sdm._memory import configure_pinned_memory +from sdm.testing import onlyCUDA + + +@pytest.mark.parametrize("version", ["2.7.0", "2.12.0", "2.12.0.dev20260701"]) +def test_pinned_memory_older_torch( + monkeypatch: pytest.MonkeyPatch, + version: str, +) -> None: + def unsupported(*args: object) -> None: + raise AssertionError("Older PyTorch must keep its allocation behavior") + + monkeypatch.setattr(torch, "__version__", version) + monkeypatch.setattr( + torch._C, + "_accelerator_getAllocatorSettings", + unsupported, + raising=False, + ) + monkeypatch.setattr( + torch._C, + "_accelerator_setAllocatorSettings", + unsupported, + raising=False, + ) + configure_pinned_memory() + + +@onlyCUDA +@pytest.mark.parametrize( + ("allocator_env", "runtime_settings", "expected_mib"), + [ + ({}, "", 6), + ( + {"PYTORCH_ALLOC_CONF": "pinned_max_cached_size_mb:64"}, + "", + 6, + ), + ( + {"PYTORCH_CUDA_ALLOC_CONF": "pinned_max_round_threshold_mb:128"}, + "", + 8, + ), + ({}, "max_split_size_mb:256,pinned_max_cached_size_mb:64", 6), + ({}, "pinned_max_round_threshold_mb:128", 8), + ], +) +def test_fit_pinned_memory( + allocator_env: dict[str, str], + runtime_settings: str, + expected_mib: int, +) -> None: + if tuple(int(part) for part in torch.__version__.split(".")[:2]) < (2, 13): + pytest.skip("Pinned allocation rounding requires PyTorch 2.13") + + # A fresh process isolates allocator state and previously retained blocks. + script = """ +import json +import sys +import torch +from sdm import Recipe, Stype +from sdm.models import ICLModel + +class CacheModel(ICLModel): + supported_feature_stypes = frozenset({Stype.numerical}) + supported_target_stypes = frozenset({Stype.numerical}) + supports_multi_target = False + supports_related_tables = False + + @classmethod + def default_recipe(cls): + return Recipe() + + def _forward(self, x_context, cache, **kwargs): + cache['tensor'] = torch.full( + (3 * 1024**2 // 4,), 7.0, device=x_context.device + ) + +x = torch.zeros(2, 1, device='cuda') +if sys.argv[1]: + torch._C._accelerator_setAllocatorSettings(sys.argv[1]) +before_settings = torch._C._accelerator_getAllocatorSettings() +before_bytes = torch.cuda.host_memory_stats()['allocated_bytes.current'] +model = CacheModel(task='regression').eval() +model.fit(x, x, num_estimators=2) +after_bytes = torch.cuda.host_memory_stats()['allocated_bytes.current'] +states = [model._cache[i]['tensor'] for i in range(2)] +print(json.dumps({ + 'before_settings': before_settings, + 'settings': torch._C._accelerator_getAllocatorSettings(), + 'bytes': after_bytes - before_bytes, + 'values_ok': all(t.is_pinned() and t[0] == 7 and t[-1] == 7 + for t in states), +})) +""" + env = { + key: value + for key, value in os.environ.items() + if key + not in { + "PYTORCH_ALLOC_CONF", + "PYTORCH_CUDA_ALLOC_CONF", + "PYTORCH_HIP_ALLOC_CONF", + } + } + env.update(allocator_env) + output = subprocess.check_output( + [sys.executable, "-c", script, runtime_settings], + env=env, + text=True, + ) + result = json.loads(output) + assert result["values_ok"] + assert ( + expected_mib * 1024**2 + <= result["bytes"] + < (expected_mib * 1024**2 + 1024) + ) + assert result["before_settings"] in result["settings"] + if "pinned_max_round_threshold_mb" not in result["before_settings"]: + assert "pinned_max_round_threshold_mb:1" in result["settings"] From 0798b566c6071002b3052c0b922bb3977e8a1a4e Mon Sep 17 00:00:00 2001 From: Lei Sun Date: Wed, 7 Oct 2026 16:50:04 +0000 Subject: [PATCH 3/3] Use PyTorch version comparison for pinned allocation support --- sdm/_memory.py | 4 ++-- test/test_memory.py | 3 ++- 2 files changed, 4 insertions(+), 3 deletions(-) diff --git a/sdm/_memory.py b/sdm/_memory.py index 1d4198a6f..94dfc6989 100644 --- a/sdm/_memory.py +++ b/sdm/_memory.py @@ -4,6 +4,7 @@ import os import torch +from torch.torch_version import TorchVersion def configure_pinned_memory() -> None: @@ -12,8 +13,7 @@ def configure_pinned_memory() -> None: The allocator setting applies process-wide and retains freed blocks for reuse. PyTorch versions before 2.13 do not support the rounding threshold. """ - version = tuple(int(part) for part in torch.__version__.split(".")[:2]) - if version < (2, 13): + if TorchVersion(torch.__version__) < "2.13": return settings = torch._C._accelerator_getAllocatorSettings() if "pinned_max_round_threshold_mb" in settings: diff --git a/test/test_memory.py b/test/test_memory.py index 554c791d8..ef1f7130e 100644 --- a/test/test_memory.py +++ b/test/test_memory.py @@ -8,6 +8,7 @@ import pytest import torch +from torch.torch_version import TorchVersion from sdm._memory import configure_pinned_memory from sdm.testing import onlyCUDA @@ -61,7 +62,7 @@ def test_fit_pinned_memory( runtime_settings: str, expected_mib: int, ) -> None: - if tuple(int(part) for part in torch.__version__.split(".")[:2]) < (2, 13): + if TorchVersion(torch.__version__) < "2.13": pytest.skip("Pinned allocation rounding requires PyTorch 2.13") # A fresh process isolates allocator state and previously retained blocks.