diff --git a/docs/source/icl.md b/docs/source/icl.md index fbab8c151..84a3a3002 100644 --- a/docs/source/icl.md +++ b/docs/source/icl.md @@ -73,6 +73,38 @@ Use one-shot {py:meth}`~sdm.models.ICLModel.forward` calls for one-time calls wh Unlike {py:meth}`~sdm.models.ICLModel.forward`, {py:meth}`~sdm.models.ICLModel.predict` does not support gradient-based fine-tuning and raises if the model is in train mode. +### Pinned host memory for fitted caches + +For CUDA fits with multiple estimators, SDM can store fitted state in pinned host memory, which supports asynchronous transfers to the GPU. +PyTorch normally rounds individual pinned allocations up to a power of two. +For example, a 3 MiB tensor can occupy a 4 MiB allocation. + +With PyTorch 2.13 or later, SDM automatically sets `pinned_max_round_threshold_mb:1` before offloading a fitted cache. +Allocations above 1 MiB then use exact sizes. +SDM preserves the current allocator options, including settings applied after CUDA initialization. +An explicit rounding threshold takes precedence over SDM's default. +Older PyTorch versions keep their existing allocation behavior. + +To select a different threshold, configure PyTorch before starting Python: + +```bash +PYTORCH_ALLOC_CONF=pinned_max_round_threshold_mb:128 python inference.py +``` + +Replace `inference.py` with your inference script. +If you already configure the allocator, add or edit the threshold in that configuration while keeping your other options. +Older PyTorch versions can reject this option. + +The setting applies to all pinned allocations in the process. +It changes allocation capacity without changing tensor contents or model arithmetic. +Exact sizes can reduce buffer reuse when allocation sizes vary, so compare fit time, prediction time, and host memory on your workload. + +This setting does not limit how much freed pinned memory PyTorch retains for reuse. +Keep `pinned_max_cached_size_mb` unchanged when evaluating rounding alone. +Reducing that separate limit can require expensive pinned allocations during later fits. +The live cache still needs space for its full tensor payload. +See [PyTorch's pinned-memory allocator options](https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage) for details. + ## Model Concepts Structured data foundation models are not bound to a specific task type. diff --git a/sdm/_memory.py b/sdm/_memory.py index af8828c47..94dfc6989 100644 --- a/sdm/_memory.py +++ b/sdm/_memory.py @@ -4,6 +4,24 @@ import os import torch +from torch.torch_version import TorchVersion + + +def configure_pinned_memory() -> None: + r"""Avoid rounding large pinned allocations unless explicitly configured. + + The allocator setting applies process-wide and retains freed blocks for + reuse. PyTorch versions before 2.13 do not support the rounding threshold. + """ + if TorchVersion(torch.__version__) < "2.13": + return + settings = torch._C._accelerator_getAllocatorSettings() + if "pinned_max_round_threshold_mb" in settings: + return + setting = "pinned_max_round_threshold_mb:1" + torch._C._accelerator_setAllocatorSettings( + f"{settings},{setting}" if settings else setting + ) def chunk_memory_limit(device: torch.device) -> int: diff --git a/sdm/models/base.py b/sdm/models/base.py index 83d9bc3f5..0f6a8c4b7 100644 --- a/sdm/models/base.py +++ b/sdm/models/base.py @@ -20,6 +20,7 @@ TaskLike, ) from sdm._inference import inference_mode +from sdm._memory import configure_pinned_memory from sdm._warnings import warn_once from sdm.cache import Cache from sdm.models.callback import Callback @@ -312,6 +313,7 @@ def fit( ) if x.is_cuda and len(contexts) > 1: + configure_pinned_memory() try: # Copy to pinned CPU memory: batch_cache = batch_cache._apply_tensor( lambda tensor: torch.ops.aten._to_copy.default( diff --git a/test/test_memory.py b/test/test_memory.py new file mode 100644 index 000000000..ef1f7130e --- /dev/null +++ b/test/test_memory.py @@ -0,0 +1,133 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +import json +import os +import subprocess +import sys + +import pytest +import torch +from torch.torch_version import TorchVersion + +from sdm._memory import configure_pinned_memory +from sdm.testing import onlyCUDA + + +@pytest.mark.parametrize("version", ["2.7.0", "2.12.0", "2.12.0.dev20260701"]) +def test_pinned_memory_older_torch( + monkeypatch: pytest.MonkeyPatch, + version: str, +) -> None: + def unsupported(*args: object) -> None: + raise AssertionError("Older PyTorch must keep its allocation behavior") + + monkeypatch.setattr(torch, "__version__", version) + monkeypatch.setattr( + torch._C, + "_accelerator_getAllocatorSettings", + unsupported, + raising=False, + ) + monkeypatch.setattr( + torch._C, + "_accelerator_setAllocatorSettings", + unsupported, + raising=False, + ) + configure_pinned_memory() + + +@onlyCUDA +@pytest.mark.parametrize( + ("allocator_env", "runtime_settings", "expected_mib"), + [ + ({}, "", 6), + ( + {"PYTORCH_ALLOC_CONF": "pinned_max_cached_size_mb:64"}, + "", + 6, + ), + ( + {"PYTORCH_CUDA_ALLOC_CONF": "pinned_max_round_threshold_mb:128"}, + "", + 8, + ), + ({}, "max_split_size_mb:256,pinned_max_cached_size_mb:64", 6), + ({}, "pinned_max_round_threshold_mb:128", 8), + ], +) +def test_fit_pinned_memory( + allocator_env: dict[str, str], + runtime_settings: str, + expected_mib: int, +) -> None: + if TorchVersion(torch.__version__) < "2.13": + pytest.skip("Pinned allocation rounding requires PyTorch 2.13") + + # A fresh process isolates allocator state and previously retained blocks. + script = """ +import json +import sys +import torch +from sdm import Recipe, Stype +from sdm.models import ICLModel + +class CacheModel(ICLModel): + supported_feature_stypes = frozenset({Stype.numerical}) + supported_target_stypes = frozenset({Stype.numerical}) + supports_multi_target = False + supports_related_tables = False + + @classmethod + def default_recipe(cls): + return Recipe() + + def _forward(self, x_context, cache, **kwargs): + cache['tensor'] = torch.full( + (3 * 1024**2 // 4,), 7.0, device=x_context.device + ) + +x = torch.zeros(2, 1, device='cuda') +if sys.argv[1]: + torch._C._accelerator_setAllocatorSettings(sys.argv[1]) +before_settings = torch._C._accelerator_getAllocatorSettings() +before_bytes = torch.cuda.host_memory_stats()['allocated_bytes.current'] +model = CacheModel(task='regression').eval() +model.fit(x, x, num_estimators=2) +after_bytes = torch.cuda.host_memory_stats()['allocated_bytes.current'] +states = [model._cache[i]['tensor'] for i in range(2)] +print(json.dumps({ + 'before_settings': before_settings, + 'settings': torch._C._accelerator_getAllocatorSettings(), + 'bytes': after_bytes - before_bytes, + 'values_ok': all(t.is_pinned() and t[0] == 7 and t[-1] == 7 + for t in states), +})) +""" + env = { + key: value + for key, value in os.environ.items() + if key + not in { + "PYTORCH_ALLOC_CONF", + "PYTORCH_CUDA_ALLOC_CONF", + "PYTORCH_HIP_ALLOC_CONF", + } + } + env.update(allocator_env) + output = subprocess.check_output( + [sys.executable, "-c", script, runtime_settings], + env=env, + text=True, + ) + result = json.loads(output) + assert result["values_ok"] + assert ( + expected_mib * 1024**2 + <= result["bytes"] + < (expected_mib * 1024**2 + 1024) + ) + assert result["before_settings"] in result["settings"] + if "pinned_max_round_threshold_mb" not in result["before_settings"]: + assert "pinned_max_round_threshold_mb:1" in result["settings"]