Uh oh!
There was an error while loading. Please reload this page.
- Notifications
You must be signed in to change notification settings - Fork 321
Switch to use CUDA driver APIs in Device constructor#460
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
Merged
Uh oh!
There was an error while loading. Please reload this page.
Merged
Changes from all commits
Commits
Show all changes
15 commits
Select commit
Hold shift + click to select a range
2afcb20
cache cc to speed it up
leofang 87405ad
avoid using cudart APIs in Device constructor
leofang 95777c4
avoid silly, redundant lock
leofang 4cfd505
Merge branch 'main' into cache_cc
leofang 7852459
Merge branch 'cache_cc' into reduce_cudart
leofang 7f11565
Merge branch 'main' into reduce_cudart
leofang 3985435
Merge branch 'main' into reduce_cudart
leofang c9fac0b
minor perf opt: try-except + skip assert
leofang 6b245ff
also optimize for explicit dev id
leofang a9e6552
Merge branch 'main' into reduce_cudart
leofang d70ec24
update release notes
leofang d279e50
debug sanitizer
leofang eb82b80
Merge branch 'main' into reduce_cudart
leofang 708fd70
fix type hint; compare against enums
leofang 4015f9c
Revert "debug sanitizer"
leofang File filter
Filter by extension
Conversations
Failed to load comments.
Loading
Uh oh!
There was an error while loading. Please reload this page.
Jump to
Jump to file
Failed to load files.
Loading
Uh oh!
There was an error while loading. Please reload this page.
Diff view
Diff view
There are no files selected for viewing
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -13,6 +13,7 @@ | ||
| from cuda.core.experimental._utils.cuda_utils import ( | ||
| ComputeCapability, | ||
| CUDAError, | ||
| _check_driver_error, | ||
| driver, | ||
| handle_return, | ||
| precondition, | ||
| @@ -930,6 +931,10 @@ def multicast_supported(self) -> bool: | ||
| return bool(self._get_cached_attribute(driver.CUdevice_attribute.CU_DEVICE_ATTRIBUTE_MULTICAST_SUPPORTED)) | ||
| _SUCCESS = driver.CUresult.CUDA_SUCCESS | ||
| _INVALID_CTX = driver.CUresult.CUDA_ERROR_INVALID_CONTEXT | ||
| class Device: | ||
| """Represent a GPU and act as an entry point for cuda.core features. | ||
| @@ -959,7 +964,7 @@ class Device: | ||
| __slots__ = ("_id", "_mr", "_has_inited", "_properties") | ||
| def __new__(cls, device_id=None): | ||
| def __new__(cls, device_id: Optional[int] = None): | ||
| global _is_cuInit | ||
| if _is_cuInit is False: | ||
| with _lock: | ||
| @@ -968,26 +973,34 @@ def __new__(cls, device_id=None): | ||
| # important: creating a Device instance does not initialize the GPU! | ||
| if device_id is None: | ||
| device_id = handle_return(runtime.cudaGetDevice()) | ||
| assert_type(device_id, int) | ||
| else: | ||
| total = handle_return(runtime.cudaGetDeviceCount()) | ||
| assert_type(device_id, int) | ||
| if not (0 <= device_id < total): | ||
| raise ValueError(f"device_id must be within [0, {total}), got {device_id}") | ||
| err, dev = driver.cuCtxGetDevice() | ||
| if err == _SUCCESS: | ||
| device_id = int(dev) | ||
| elif err == _INVALID_CTX: | ||
| ctx = handle_return(driver.cuCtxGetCurrent()) | ||
| assert int(ctx) == 0 | ||
| device_id = 0 # cudart behavior | ||
| else: | ||
| _check_driver_error(err) | ||
| elif device_id < 0: | ||
| raise ValueError(f"device_id must be >= 0, got {device_id}") | ||
| # ensure Device is singleton | ||
| if not hasattr(_tls, "devices"): | ||
| total = handle_return(runtime.cudaGetDeviceCount()) | ||
| _tls.devices = [] | ||
| try: | ||
| devices = _tls.devices | ||
| except AttributeError: | ||
| total = handle_return(driver.cuDeviceGetCount()) | ||
leofang marked this conversation as resolved.
Uh oh!There was an error while loading. Please reload this page. | ||
| devices = _tls.devices = [] | ||
| for dev_id in range(total): | ||
leofang marked this conversation as resolved.
Uh oh!There was an error while loading. Please reload this page. | ||
| dev = super().__new__(cls) | ||
| dev._id = dev_id | ||
| # If the device is in TCC mode, or does not support memory pools for some other reason, | ||
| # use the SynchronousMemoryResource which does not use memory pools. | ||
| if ( | ||
| handle_return( | ||
| runtime.cudaDeviceGetAttribute(runtime.cudaDeviceAttr.cudaDevAttrMemoryPoolsSupported, 0) | ||
| driver.cuDeviceGetAttribute( | ||
| driver.CUdevice_attribute.CU_DEVICE_ATTRIBUTE_MEMORY_POOLS_SUPPORTED, dev_id | ||
| ) | ||
| ) | ||
| ) == 1: | ||
| dev._mr = _DefaultAsyncMempool(dev_id) | ||
| @@ -996,9 +1009,12 @@ def __new__(cls, device_id=None): | ||
| dev._has_inited = False | ||
| dev._properties = None | ||
| _tls.devices.append(dev) | ||
| devices.append(dev) | ||
| return _tls.devices[device_id] | ||
| try: | ||
| return devices[device_id] | ||
| except IndexError: | ||
| raise ValueError(f"device_id must be within [0, {len(devices)}), got {device_id}") from None | ||
| def _check_context_initialized(self, *args, **kwargs): | ||
| if not self._has_inited: | ||
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
Add this suggestion to a batch that can be applied as a single commit.This suggestion is invalid because no changes were made to the code.Suggestions cannot be applied while the pull request is closed.Suggestions cannot be applied while viewing a subset of changes.Only one suggestion per line can be applied in a batch.Add this suggestion to a batch that can be applied as a single commit.Applying suggestions on deleted lines is not supported.You must change the existing code in this line in order to create a valid suggestion.Outdated suggestions cannot be applied.This suggestion has been applied or marked resolved.Suggestions cannot be applied from pending reviews.Suggestions cannot be applied on multi-line comments.Suggestions cannot be applied while the pull request is queued to merge.Suggestion cannot be applied right now. Please check back later.
Uh oh!
There was an error while loading. Please reload this page.