Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
32 changes: 32 additions & 0 deletions docs/source/icl.md
Original file line number Diff line number Diff line change
Expand Up @@ -73,6 +73,38 @@ Use one-shot {py:meth}`~sdm.models.ICLModel.forward` calls for one-time calls wh

Unlike {py:meth}`~sdm.models.ICLModel.forward`, {py:meth}`~sdm.models.ICLModel.predict` does not support gradient-based fine-tuning and raises if the model is in train mode.

### Pinned host memory for fitted caches

For CUDA fits with multiple estimators, SDM can store fitted state in pinned host memory, which supports asynchronous transfers to the GPU.
PyTorch normally rounds individual pinned allocations up to a power of two.
For example, a 3 MiB tensor can occupy a 4 MiB allocation.

With PyTorch 2.13 or later, SDM automatically sets `pinned_max_round_threshold_mb:1` before offloading a fitted cache.
Allocations above 1 MiB then use exact sizes.
SDM preserves the current allocator options, including settings applied after CUDA initialization.
An explicit rounding threshold takes precedence over SDM's default.
Older PyTorch versions keep their existing allocation behavior.

To select a different threshold, configure PyTorch before starting Python:

```bash
PYTORCH_ALLOC_CONF=pinned_max_round_threshold_mb:128 python inference.py
```

Replace `inference.py` with your inference script.
If you already configure the allocator, add or edit the threshold in that configuration while keeping your other options.
Older PyTorch versions can reject this option.

The setting applies to all pinned allocations in the process.
It changes allocation capacity without changing tensor contents or model arithmetic.
Exact sizes can reduce buffer reuse when allocation sizes vary, so compare fit time, prediction time, and host memory on your workload.

This setting does not limit how much freed pinned memory PyTorch retains for reuse.
Keep `pinned_max_cached_size_mb` unchanged when evaluating rounding alone.
Reducing that separate limit can require expensive pinned allocations during later fits.
The live cache still needs space for its full tensor payload.
See [PyTorch's pinned-memory allocator options](https://docs.pytorch.org/docs/stable/notes/cuda.html#optimizing-memory-usage) for details.

## Model Concepts

Structured data foundation models are not bound to a specific task type.
Expand Down
18 changes: 18 additions & 0 deletions sdm/_memory.py
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,24 @@
import os

import torch
from torch.torch_version import TorchVersion


def configure_pinned_memory() -> None:
r"""Avoid rounding large pinned allocations unless explicitly configured.

The allocator setting applies process-wide and retains freed blocks for
reuse. PyTorch versions before 2.13 do not support the rounding threshold.
"""
if TorchVersion(torch.__version__) < "2.13":
return
settings = torch._C._accelerator_getAllocatorSettings()
if "pinned_max_round_threshold_mb" in settings:
return
setting = "pinned_max_round_threshold_mb:1"
torch._C._accelerator_setAllocatorSettings(
f"{settings},{setting}" if settings else setting
)


def chunk_memory_limit(device: torch.device) -> int:
Expand Down
2 changes: 2 additions & 0 deletions sdm/models/base.py
Original file line number Diff line number Diff line change
Expand Up @@ -20,6 +20,7 @@
TaskLike,
)
from sdm._inference import inference_mode
from sdm._memory import configure_pinned_memory
from sdm._warnings import warn_once
from sdm.cache import Cache
from sdm.models.callback import Callback
Expand Down Expand Up @@ -312,6 +313,7 @@ def fit(
)

if x.is_cuda and len(contexts) > 1:
configure_pinned_memory()
try: # Copy to pinned CPU memory:
batch_cache = batch_cache._apply_tensor(
lambda tensor: torch.ops.aten._to_copy.default(
Expand Down
133 changes: 133 additions & 0 deletions test/test_memory.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,133 @@
# SPDX-FileCopyrightText: Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: Apache-2.0

import json
import os
import subprocess
import sys

import pytest
import torch
from torch.torch_version import TorchVersion

from sdm._memory import configure_pinned_memory
from sdm.testing import onlyCUDA


@pytest.mark.parametrize("version", ["2.7.0", "2.12.0", "2.12.0.dev20260701"])
def test_pinned_memory_older_torch(
monkeypatch: pytest.MonkeyPatch,
version: str,
) -> None:
def unsupported(*args: object) -> None:
raise AssertionError("Older PyTorch must keep its allocation behavior")

monkeypatch.setattr(torch, "__version__", version)
monkeypatch.setattr(
torch._C,
"_accelerator_getAllocatorSettings",
unsupported,
raising=False,
)
monkeypatch.setattr(
torch._C,
"_accelerator_setAllocatorSettings",
unsupported,
raising=False,
)
configure_pinned_memory()


@onlyCUDA
@pytest.mark.parametrize(
("allocator_env", "runtime_settings", "expected_mib"),
[
({}, "", 6),
(
{"PYTORCH_ALLOC_CONF": "pinned_max_cached_size_mb:64"},
"",
6,
),
(
{"PYTORCH_CUDA_ALLOC_CONF": "pinned_max_round_threshold_mb:128"},
"",
8,
),
({}, "max_split_size_mb:256,pinned_max_cached_size_mb:64", 6),
({}, "pinned_max_round_threshold_mb:128", 8),
],
)
def test_fit_pinned_memory(
allocator_env: dict[str, str],
runtime_settings: str,
expected_mib: int,
) -> None:
if TorchVersion(torch.__version__) < "2.13":
pytest.skip("Pinned allocation rounding requires PyTorch 2.13")

# A fresh process isolates allocator state and previously retained blocks.
script = """
import json
import sys
import torch
from sdm import Recipe, Stype
from sdm.models import ICLModel

class CacheModel(ICLModel):
supported_feature_stypes = frozenset({Stype.numerical})
supported_target_stypes = frozenset({Stype.numerical})
supports_multi_target = False
supports_related_tables = False

@classmethod
def default_recipe(cls):
return Recipe()

def _forward(self, x_context, cache, **kwargs):
cache['tensor'] = torch.full(
(3 * 1024**2 // 4,), 7.0, device=x_context.device
)

x = torch.zeros(2, 1, device='cuda')
if sys.argv[1]:
torch._C._accelerator_setAllocatorSettings(sys.argv[1])
before_settings = torch._C._accelerator_getAllocatorSettings()
before_bytes = torch.cuda.host_memory_stats()['allocated_bytes.current']
model = CacheModel(task='regression').eval()
model.fit(x, x, num_estimators=2)
after_bytes = torch.cuda.host_memory_stats()['allocated_bytes.current']
states = [model._cache[i]['tensor'] for i in range(2)]
print(json.dumps({
'before_settings': before_settings,
'settings': torch._C._accelerator_getAllocatorSettings(),
'bytes': after_bytes - before_bytes,
'values_ok': all(t.is_pinned() and t[0] == 7 and t[-1] == 7
for t in states),
}))
"""
env = {
key: value
for key, value in os.environ.items()
if key
not in {
"PYTORCH_ALLOC_CONF",
"PYTORCH_CUDA_ALLOC_CONF",
"PYTORCH_HIP_ALLOC_CONF",
}
}
env.update(allocator_env)
output = subprocess.check_output(
[sys.executable, "-c", script, runtime_settings],
env=env,
text=True,
)
result = json.loads(output)
assert result["values_ok"]
assert (
expected_mib * 1024**2
<= result["bytes"]
< (expected_mib * 1024**2 + 1024)
)
assert result["before_settings"] in result["settings"]
if "pinned_max_round_threshold_mb" not in result["before_settings"]:
assert "pinned_max_round_threshold_mb:1" in result["settings"]
Loading