|
|
API Reference Manual
|
January
2024
Table of Contents
Chapter 1. Difference between the driver and runtime APIs
1
Chapter 2. API synchronization behavior
3
Chapter 3. Stream synchronization behavior
5
Chapter 4. Graph object thread safety
7
Chapter 5. Rules for version mixing
8
Chapter 6. Modules
9
6.1. Data types used by CUDA driver
10
CUaccessPolicyWindow_v1
11
CUarrayMapInfo_v1
11
CUasyncNotificationInfo
11
CUcheckpointCheckpointArgs
11
CUcheckpointLockArgs
11
CUcheckpointRestoreArgs
11
CUcheckpointUnlockArgs
11
CUctxCigParam
11
CUctxCreateParams
11
CUDA_ARRAY3D_DESCRIPTOR_v2
11
CUDA_ARRAY_DESCRIPTOR_v2
11
CUDA_ARRAY_MEMORY_REQUIREMENTS_v1
11
CUDA_ARRAY_SPARSE_PROPERTIES_v1
11
CUDA_BATCH_MEM_OP_NODE_PARAMS_v2
11
CUDA_CHILD_GRAPH_NODE_PARAMS
11
CUDA_CONDITIONAL_NODE_PARAMS
11
CUDA_EVENT_RECORD_NODE_PARAMS
11
CUDA_EVENT_WAIT_NODE_PARAMS
12
CUDA_EXT_SEM_SIGNAL_NODE_PARAMS_v1
12
CUDA_EXT_SEM_SIGNAL_NODE_PARAMS_v2
12
CUDA_EXT_SEM_WAIT_NODE_PARAMS_v1
12
CUDA_EXT_SEM_WAIT_NODE_PARAMS_v2
12
CUDA_EXTERNAL_MEMORY_BUFFER_DESC_v1
12
CUDA_EXTERNAL_MEMORY_HANDLE_DESC_v1
12
CUDA_EXTERNAL_MEMORY_MIPMAPPED_ARRAY_DESC_v1
12
CUDA_EXTERNAL_SEMAPHORE_HANDLE_DESC_v1
12
CUDA_EXTERNAL_SEMAPHORE_SIGNAL_PARAMS_v1
12
|
ii
CUDA_EXTERNAL_SEMAPHORE_WAIT_PARAMS_v1
12
CUDA_GRAPH_INSTANTIATE_PARAMS
12
CUDA_HOST_NODE_PARAMS_v1
12
CUDA_HOST_NODE_PARAMS_v2
13
CUDA_KERNEL_NODE_PARAMS_v1
13
CUDA_KERNEL_NODE_PARAMS_v2
13
CUDA_KERNEL_NODE_PARAMS_v3
13
CUDA_LAUNCH_PARAMS_v1
13
CUDA_MEM_ALLOC_NODE_PARAMS_v1
13
CUDA_MEM_ALLOC_NODE_PARAMS_v2
13
CUDA_MEM_FREE_NODE_PARAMS
13
CUDA_MEMCPY2D_v2
13
CUDA_MEMCPY3D_PEER_v1
13
CUDA_MEMCPY3D_v2
13
CUDA_MEMCPY_NODE_PARAMS
13
CUDA_MEMSET_NODE_PARAMS_v1
13
CUDA_MEMSET_NODE_PARAMS_v2
13
CUDA_POINTER_ATTRIBUTE_P2P_TOKENS_v1
13
CUDA_RESOURCE_DESC_v1
13
CUDA_RESOURCE_VIEW_DESC_v1
14
CUDA_TEXTURE_DESC_v1
14
CUdevprop_v1
14
CUeglFrame_v1
14
CUexecAffinityParam_v1
14
CUexecAffinitySmCount_v1
14
CUextent3D_v1
14
CUgraphEdgeData
14
CUgraphExecUpdateResultInfo_v1
14
CUgraphNodeParams
14
CUipcEventHandle_v1
14
CUipcMemHandle_v1
14
CUlaunchAttribute
14
CUlaunchAttributeValue
14
CUlaunchConfig
14
CUlaunchMemSyncDomainMap
14
CUmemAccessDesc_v1
14
CUmemAllocationProp_v1
15
CUmemcpy3DOperand_v1
15
|
iii
CUmemcpyAttributes_v1
15
CUmemFabricHandle_v1
15
CUmemLocation_v1
15
CUmemPoolProps_v1
15
CUmemPoolPtrExportData_v1
15
CUmulticastObjectProp_v1
15
CUoffset3D_v1
15
CUstreamBatchMemOpParams_v1
15
CUtensorMap
15
cl_context_flags
15
cl_event_flags
16
CUaccessProperty
16
CUaddress_mode
16
CUarray_cubemap_face
17
CUarray_format
17
CUarraySparseSubresourceType
20
CUasyncNotificationType
20
CUclusterSchedulingPolicy
20
CUcomputemode
20
CUctx_flags
21
CUDA_POINTER_ATTRIBUTE_ACCESS_FLAGS
21
CUdevice_attribute
22
CUdevice_P2PAttribute
29
CUdeviceNumaConfig
30
CUdriverProcAddress_flags
30
CUdriverProcAddressQueryResult
30
CUeglColorFormat
31
CUeglFrameType
37
CUeglResourceLocationFlags
38
CUevent_flags
38
CUevent_record_flags
39
CUevent_sched_flags
39
CUevent_wait_flags
39
CUexecAffinityType
39
CUexternalMemoryHandleType
40
CUexternalSemaphoreHandleType
40
CUfilter_mode
41
CUflushGPUDirectRDMAWritesOptions
41
|
iv
CUflushGPUDirectRDMAWritesScope
41
CUflushGPUDirectRDMAWritesTarget
41
CUfunc_cache
42
CUfunction_attribute
42
CUGPUDirectRDMAWritesOrdering
44
CUgraphChildGraphNodeOwnership
44
CUgraphConditionalNodeType
44
CUgraphDebugDot_flags
45
CUgraphDependencyType
46
CUgraphExecUpdateResult
46
CUgraphicsMapResourceFlags
46
CUgraphicsRegisterFlags
47
CUgraphInstantiate_flags
47
CUgraphInstantiateResult
47
CUgraphNodeType
48
CUipcMem_flags
49
CUjit_cacheMode
49
CUjit_fallback
49
CUjit_option
49
CUjit_target
53
CUjitInputType
55
CUlaunchAttributeID
56
CUlaunchMemSyncDomain
58
CUlibraryOption
59
CUlimit
59
CUmem_advise
60
CUmemAccess_flags
60
CUmemAllocationCompType
61
CUmemAllocationGranularity_flags
61
CUmemAllocationHandleType
61
CUmemAllocationType
61
CUmemAttach_flags
62
CUmemcpy3DOperandType
62
CUmemcpyFlags
62
CUmemcpySrcAccessOrder
63
CUmemHandleType
63
CUmemLocationType
63
CUmemOperationType
64
|
v
CUmemorytype
64
CUmemPool_attribute
64
CUmemRangeFlags
65
CUmemRangeHandleType
65
CUmulticastGranularity_flags
65
CUoccupancy_flags
66
CUpointer_attribute
66
CUprocessState
67
CUresourcetype
67
CUresourceViewFormat
68
CUresult
69
CUshared_carveout
77
CUsharedconfig
77
CUstream_flags
77
CUstreamBatchMemOpType
78
CUstreamCaptureMode
78
CUstreamCaptureStatus
78
CUstreamMemoryBarrier_flags
79
CUstreamUpdateCaptureDependencies_flags
79
CUstreamWaitValue_flags
79
CUstreamWriteValue_flags
80
CUtensorMapDataType
80
CUtensorMapFloatOOBfill
80
CUtensorMapIm2ColWideMode
81
CUtensorMapInterleave
81
CUtensorMapL2promotion
81
CUtensorMapSwizzle
81
CUuserObject_flags
82
CUuserObjectRetain_flags
82
CUaccessPolicyWindow
82
CUarray
82
CUasyncCallback
82
CUasyncCallbackHandle
82
CUcontext
82
CUdevice
82
CUdevice_v1
83
CUdeviceptr
83
CUdeviceptr_v2
83
|
vi
CUeglStreamConnection
83
CUevent
83
CUexecAffinityParam
83
CUexternalMemory
83
CUexternalSemaphore
83
CUfunction
83
CUgraph
83
CUgraphConditionalHandle
84
CUgraphDeviceNode
84
CUgraphExec
84
CUgraphicsResource
84
CUgraphNode
84
CUgreenCtx
84
CUhostFn
84
CUkernel
84
CUlibrary
84
CUmemoryPool
84
CUmipmappedArray
85
CUmodule
85
CUoccupancyB2DSize
85
CUstream
85
CUstreamCallback
85
CUsurfObject
85
CUsurfObject_v1
85
CUsurfref
85
CUtexObject
85
CUtexObject_v1
86
CUtexref
86
CUuserObject
86
CU_ARRAY_SPARSE_PROPERTIES_SINGLE_MIPTAIL
86
CU_DEVICE_CPU
86
CU_DEVICE_INVALID
86
CU_GRAPH_COND_ASSIGN_DEFAULT
86
CU_GRAPH_KERNEL_NODE_PORT_DEFAULT
86
CU_GRAPH_KERNEL_NODE_PORT_LAUNCH_ORDER
87
CU_GRAPH_KERNEL_NODE_PORT_PROGRAMMATIC
87
CU_IPC_HANDLE_SIZE
87
CU_LAUNCH_KERNEL_REQUIRED_BLOCK_DIM
87
|
vii
CU_LAUNCH_PARAM_BUFFER_POINTER
87
CU_LAUNCH_PARAM_BUFFER_POINTER_AS_INT
88
CU_LAUNCH_PARAM_BUFFER_SIZE
88
CU_LAUNCH_PARAM_BUFFER_SIZE_AS_INT
88
CU_LAUNCH_PARAM_END
88
CU_LAUNCH_PARAM_END_AS_INT
88
CU_MEM_CREATE_USAGE_HW_DECOMPRESS
88
CU_MEM_CREATE_USAGE_TILE_POOL
88
CU_MEM_POOL_CREATE_USAGE_HW_DECOMPRESS
89
CU_MEMHOSTALLOC_DEVICEMAP
89
CU_MEMHOSTALLOC_PORTABLE
89
CU_MEMHOSTALLOC_WRITECOMBINED
89
CU_MEMHOSTREGISTER_DEVICEMAP
89
CU_MEMHOSTREGISTER_IOMEMORY
89
CU_MEMHOSTREGISTER_PORTABLE
89
CU_MEMHOSTREGISTER_READ_ONLY
90
CU_PARAM_TR_DEFAULT
90
CU_STREAM_LEGACY
90
CU_STREAM_PER_THREAD
90
CU_TENSOR_MAP_NUM_QWORDS
90
CU_TRSA_OVERRIDE_FORMAT
90
CU_TRSF_DISABLE_TRILINEAR_OPTIMIZATION
91
CU_TRSF_NORMALIZED_COORDINATES
91
CU_TRSF_READ_AS_INTEGER
91
CU_TRSF_SEAMLESS_CUBEMAP
91
CU_TRSF_SRGB
91
CUDA_ARRAY3D_2DARRAY
91
CUDA_ARRAY3D_COLOR_ATTACHMENT
91
CUDA_ARRAY3D_CUBEMAP
91
CUDA_ARRAY3D_DEFERRED_MAPPING
92
CUDA_ARRAY3D_DEPTH_TEXTURE
92
CUDA_ARRAY3D_LAYERED
92
CUDA_ARRAY3D_SPARSE
92
CUDA_ARRAY3D_SURFACE_LDST
92
CUDA_ARRAY3D_TEXTURE_GATHER
92
CUDA_ARRAY3D_VIDEO_ENCODE_DECODE
92
CUDA_COOPERATIVE_LAUNCH_MULTI_DEVICE_NO_POST_LAUNCH_SYNC
93
CUDA_COOPERATIVE_LAUNCH_MULTI_DEVICE_NO_PRE_LAUNCH_SYNC
93
|
viii
CUDA_EGL_INFINITE_TIMEOUT
93
CUDA_EXTERNAL_MEMORY_DEDICATED
93
CUDA_EXTERNAL_SEMAPHORE_SIGNAL_SKIP_NVSCIBUF_MEMSYNC
93
CUDA_EXTERNAL_SEMAPHORE_WAIT_SKIP_NVSCIBUF_MEMSYNC
94
CUDA_NVSCISYNC_ATTR_SIGNAL
94
CUDA_NVSCISYNC_ATTR_WAIT
94
CUDA_VERSION
94
MAX_PLANES
94
6.2. Error Handling
94
cuGetErrorName
95
cuGetErrorString
95
6.3. Initialization
96
cuInit
96
6.4. Version Management
96
cuDriverGetVersion
97
6.5. Device Management
97
cuDeviceGet
97
cuDeviceGetAttribute
98
cuDeviceGetCount
105
cuDeviceGetDefaultMemPool
105
cuDeviceGetExecAffinitySupport
106
cuDeviceGetLuid
107
cuDeviceGetMemPool
107
cuDeviceGetName
108
cuDeviceGetNvSciSyncAttributes
109
cuDeviceGetTexture1DLinearMaxWidth
110
cuDeviceGetUuid
111
cuDeviceGetUuid_v2
112
cuDeviceSetMemPool
112
cuDeviceTotalMem
113
cuFlushGPUDirectRDMAWrites
114
6.6. Device Management [DEPRECATED]
115
cuDeviceComputeCapability
115
cuDeviceGetProperties
116
6.7. Primary Context Management
117
cuDevicePrimaryCtxGetState
117
cuDevicePrimaryCtxRelease
118
cuDevicePrimaryCtxReset
119
|
ix
cuDevicePrimaryCtxRetain
119
cuDevicePrimaryCtxSetFlags
120
6.8. Context Management
122
cuCtxCreate
123
cuCtxCreate_v3
125
cuCtxCreate_v4
128
cuCtxDestroy
131
cuCtxGetApiVersion
132
cuCtxGetCacheConfig
133
cuCtxGetCurrent
134
cuCtxGetDevice
134
cuCtxGetExecAffinity
135
cuCtxGetFlags
135
cuCtxGetId
136
cuCtxGetLimit
137
cuCtxGetStreamPriorityRange
138
cuCtxPopCurrent
139
cuCtxPushCurrent
139
cuCtxRecordEvent
140
cuCtxResetPersistingL2Cache
141
cuCtxSetCacheConfig
141
cuCtxSetCurrent
142
cuCtxSetFlags
143
cuCtxSetLimit
144
cuCtxSynchronize
145
cuCtxWaitEvent
146
6.9. Context Management [DEPRECATED]
147
cuCtxAttach
147
cuCtxDetach
148
cuCtxGetSharedMemConfig
149
cuCtxSetSharedMemConfig
150
6.10. Module Management
151
CUmoduleLoadingMode
151
cuLinkAddData
151
cuLinkAddFile
152
cuLinkComplete
153
cuLinkCreate
154
cuLinkDestroy
155
|
x
cuModuleEnumerateFunctions
155
cuModuleGetFunction
156
cuModuleGetFunctionCount
157
cuModuleGetGlobal
157
cuModuleGetLoadingMode
158
cuModuleLoad
159
cuModuleLoadData
160
cuModuleLoadDataEx
161
cuModuleLoadFatBinary
162
cuModuleUnload
163
6.11. Module Management [DEPRECATED]
163
cuModuleGetSurfRef
164
cuModuleGetTexRef
164
6.12. Library Management
165
cuKernelGetAttribute
165
cuKernelGetFunction
167
cuKernelGetLibrary
168
cuKernelGetName
169
cuKernelGetParamInfo
169
cuKernelSetAttribute
170
cuKernelSetCacheConfig
172
cuLibraryEnumerateKernels
173
cuLibraryGetGlobal
174
cuLibraryGetKernel
174
cuLibraryGetKernelCount
175
cuLibraryGetManaged
176
cuLibraryGetModule
176
cuLibraryGetUnifiedFunction
177
cuLibraryLoadData
178
cuLibraryLoadFromFile
179
cuLibraryUnload
181
6.13. Memory Management
181
CUmemDecompressParams
182
CUmemDecompressAlgorithm
182
cuArray3DCreate
182
cuArray3DGetDescriptor
186
cuArrayCreate
187
cuArrayDestroy
189
|
xi
cuArrayGetDescriptor
190
cuArrayGetMemoryRequirements
191
cuArrayGetPlane
192
cuArrayGetSparseProperties
193
cuDeviceGetByPCIBusId
194
cuDeviceGetPCIBusId
194
cuDeviceRegisterAsyncNotification
195
cuDeviceUnregisterAsyncNotification
196
cuIpcCloseMemHandle
197
cuIpcGetEventHandle
198
cuIpcGetMemHandle
199
cuIpcOpenEventHandle
199
cuIpcOpenMemHandle
200
cuMemAlloc
202
cuMemAllocHost
202
cuMemAllocManaged
204
cuMemAllocPitch
206
cuMemBatchDecompressAsync
208
cuMemcpy
209
cuMemcpy2D
210
cuMemcpy2DAsync
213
cuMemcpy2DUnaligned
215
cuMemcpy3D
218
cuMemcpy3DAsync
221
cuMemcpy3DBatchAsync
224
cuMemcpy3DPeer
226
cuMemcpy3DPeerAsync
226
cuMemcpyAsync
227
cuMemcpyAtoA
228
cuMemcpyAtoD
229
cuMemcpyAtoH
230
cuMemcpyAtoHAsync
232
cuMemcpyBatchAsync
233
cuMemcpyDtoA
235
cuMemcpyDtoD
236
cuMemcpyDtoDAsync
237
cuMemcpyDtoH
238
cuMemcpyDtoHAsync
239
|
xii
cuMemcpyHtoA
240
cuMemcpyHtoAAsync
241
cuMemcpyHtoD
243
cuMemcpyHtoDAsync
244
cuMemcpyPeer
245
cuMemcpyPeerAsync
246
cuMemFree
247
cuMemFreeHost
248
cuMemGetAddressRange
248
cuMemGetHandleForAddressRange
249
cuMemGetInfo
250
cuMemHostAlloc
251
cuMemHostGetDevicePointer
253
cuMemHostGetFlags
255
cuMemHostRegister
255
cuMemHostUnregister
257
cuMemsetD16
258
cuMemsetD16Async
259
cuMemsetD2D16
260
cuMemsetD2D16Async
261
cuMemsetD2D32
262
cuMemsetD2D32Async
263
cuMemsetD2D8
264
cuMemsetD2D8Async
265
cuMemsetD32
267
cuMemsetD32Async
268
cuMemsetD8
269
cuMemsetD8Async
270
cuMipmappedArrayCreate
271
cuMipmappedArrayDestroy
274
cuMipmappedArrayGetLevel
275
cuMipmappedArrayGetMemoryRequirements
276
cuMipmappedArrayGetSparseProperties
277
6.14. Virtual Memory Management
277
cuMemAddressFree
278
cuMemAddressReserve
278
cuMemCreate
279
cuMemExportToShareableHandle
281
|
xiii
cuMemGetAccess
282
cuMemGetAllocationGranularity
282
cuMemGetAllocationPropertiesFromHandle
283
cuMemImportFromShareableHandle
284
cuMemMap
285
cuMemMapArrayAsync
286
cuMemRelease
289
cuMemRetainAllocationHandle
290
cuMemSetAccess
291
cuMemUnmap
292
6.15. Stream Ordered Memory Allocator
292
cuMemAllocAsync
293
cuMemAllocFromPoolAsync
294
cuMemFreeAsync
295
cuMemPoolCreate
296
cuMemPoolDestroy
297
cuMemPoolExportPointer
297
cuMemPoolExportToShareableHandle
298
cuMemPoolGetAccess
299
cuMemPoolGetAttribute
300
cuMemPoolImportFromShareableHandle
301
cuMemPoolImportPointer
302
cuMemPoolSetAccess
303
cuMemPoolSetAttribute
303
cuMemPoolTrimTo
304
6.16. Multicast Object Management
305
cuMulticastAddDevice
306
cuMulticastBindAddr
307
cuMulticastBindMem
308
cuMulticastCreate
309
cuMulticastGetGranularity
310
cuMulticastUnbind
311
6.17. Unified Addressing
312
cuMemAdvise
313
cuMemAdvise_v2
316
cuMemPrefetchAsync
320
cuMemPrefetchAsync_v2
322
cuMemRangeGetAttribute
324
|
xiv
cuMemRangeGetAttributes
326
cuPointerGetAttribute
327
cuPointerGetAttributes
331
cuPointerSetAttribute
332
6.18. Stream Management
333
cuStreamAddCallback
333
cuStreamAttachMemAsync
335
cuStreamBeginCapture
337
cuStreamBeginCaptureToGraph
338
cuStreamCopyAttributes
339
cuStreamCreate
340
cuStreamCreateWithPriority
341
cuStreamDestroy
342
cuStreamEndCapture
342
cuStreamGetAttribute
343
cuStreamGetCaptureInfo
344
cuStreamGetCaptureInfo_v3
345
cuStreamGetCtx
347
cuStreamGetCtx_v2
348
cuStreamGetDevice
349
cuStreamGetFlags
350
cuStreamGetId
350
cuStreamGetPriority
351
cuStreamIsCapturing
352
cuStreamQuery
353
cuStreamSetAttribute
354
cuStreamSynchronize
354
cuStreamUpdateCaptureDependencies
355
cuStreamUpdateCaptureDependencies_v2
356
cuStreamWaitEvent
357
cuThreadExchangeStreamCaptureMode
358
6.19. Event Management
359
cuEventCreate
359
cuEventDestroy
360
cuEventElapsedTime
361
cuEventElapsedTime_v2
362
cuEventQuery
363
cuEventRecord
364
|
xv
cuEventRecordWithFlags
365
cuEventSynchronize
366
6.20. External Resource Interoperability
366
cuDestroyExternalMemory
367
cuDestroyExternalSemaphore
367
cuExternalMemoryGetMappedBuffer
368
cuExternalMemoryGetMappedMipmappedArray
369
cuImportExternalMemory
370
cuImportExternalSemaphore
373
cuSignalExternalSemaphoresAsync
376
cuWaitExternalSemaphoresAsync
378
6.21. Stream Memory Operations
380
cuStreamBatchMemOp
381
cuStreamWaitValue32
382
cuStreamWaitValue64
383
cuStreamWriteValue32
384
cuStreamWriteValue64
385
6.22. Execution Control
386
cuFuncGetAttribute
386
cuFuncGetModule
388
cuFuncGetName
389
cuFuncGetParamInfo
389
cuFuncIsLoaded
390
cuFuncLoad
391
cuFuncSetAttribute
391
cuFuncSetCacheConfig
393
cuLaunchCooperativeKernel
394
cuLaunchCooperativeKernelMultiDevice
396
cuLaunchHostFunc
399
cuLaunchKernel
400
cuLaunchKernelEx
403
6.23. Execution Control [DEPRECATED]
407
cuFuncSetBlockShape
408
cuFuncSetSharedMemConfig
409
cuFuncSetSharedSize
410
cuLaunch
411
cuLaunchGrid
412
cuLaunchGridAsync
413
|
xvi
cuParamSetf
414
cuParamSeti
415
cuParamSetSize
415
cuParamSetTexRef
416
cuParamSetv
417
6.24. Graph Management
418
cuDeviceGetGraphMemAttribute
418
cuDeviceGraphMemTrim
419
cuDeviceSetGraphMemAttribute
419
cuGraphAddBatchMemOpNode
420
cuGraphAddChildGraphNode
421
cuGraphAddDependencies
422
cuGraphAddDependencies_v2
423
cuGraphAddEmptyNode
424
cuGraphAddEventRecordNode
425
cuGraphAddEventWaitNode
426
cuGraphAddExternalSemaphoresSignalNode
427
cuGraphAddExternalSemaphoresWaitNode
429
cuGraphAddHostNode
430
cuGraphAddKernelNode
431
cuGraphAddMemAllocNode
433
cuGraphAddMemcpyNode
435
cuGraphAddMemFreeNode
436
cuGraphAddMemsetNode
438
cuGraphAddNode
439
cuGraphAddNode_v2
440
cuGraphBatchMemOpNodeGetParams
441
cuGraphBatchMemOpNodeSetParams
442
cuGraphChildGraphNodeGetGraph
443
cuGraphClone
443
cuGraphConditionalHandleCreate
444
cuGraphCreate
445
cuGraphDebugDotPrint
446
cuGraphDestroy
447
cuGraphDestroyNode
447
cuGraphEventRecordNodeGetEvent
448
cuGraphEventRecordNodeSetEvent
449
cuGraphEventWaitNodeGetEvent
449
|
xvii
cuGraphEventWaitNodeSetEvent
450
cuGraphExecBatchMemOpNodeSetParams
451
cuGraphExecChildGraphNodeSetParams
452
cuGraphExecDestroy
453
cuGraphExecEventRecordNodeSetEvent
454
cuGraphExecEventWaitNodeSetEvent
455
cuGraphExecExternalSemaphoresSignalNodeSetParams
456
cuGraphExecExternalSemaphoresWaitNodeSetParams
457
cuGraphExecGetFlags
458
cuGraphExecHostNodeSetParams
459
cuGraphExecKernelNodeSetParams
460
cuGraphExecMemcpyNodeSetParams
461
cuGraphExecMemsetNodeSetParams
462
cuGraphExecNodeSetParams
463
cuGraphExecUpdate
465
cuGraphExternalSemaphoresSignalNodeGetParams
468
cuGraphExternalSemaphoresSignalNodeSetParams
469
cuGraphExternalSemaphoresWaitNodeGetParams
470
cuGraphExternalSemaphoresWaitNodeSetParams
471
cuGraphGetEdges
472
cuGraphGetEdges_v2
473
cuGraphGetNodes
474
cuGraphGetRootNodes
475
cuGraphHostNodeGetParams
476
cuGraphHostNodeSetParams
476
cuGraphInstantiate
477
cuGraphInstantiateWithParams
479
cuGraphKernelNodeCopyAttributes
481
cuGraphKernelNodeGetAttribute
482
cuGraphKernelNodeGetParams
483
cuGraphKernelNodeSetAttribute
484
cuGraphKernelNodeSetParams
484
cuGraphLaunch
485
cuGraphMemAllocNodeGetParams
486
cuGraphMemcpyNodeGetParams
487
cuGraphMemcpyNodeSetParams
487
cuGraphMemFreeNodeGetParams
488
cuGraphMemsetNodeGetParams
489
|
xviii
cuGraphMemsetNodeSetParams
489
cuGraphNodeFindInClone
490
cuGraphNodeGetDependencies
491
cuGraphNodeGetDependencies_v2
492
cuGraphNodeGetDependentNodes
493
cuGraphNodeGetDependentNodes_v2
494
cuGraphNodeGetEnabled
495
cuGraphNodeGetType
496
cuGraphNodeSetEnabled
496
cuGraphNodeSetParams
498
cuGraphReleaseUserObject
498
cuGraphRemoveDependencies
499
cuGraphRemoveDependencies_v2
500
cuGraphRetainUserObject
501
cuGraphUpload
502
cuUserObjectCreate
503
cuUserObjectRelease
504
cuUserObjectRetain
504
6.25. Occupancy
505
cuOccupancyAvailableDynamicSMemPerBlock
505
cuOccupancyMaxActiveBlocksPerMultiprocessor
506
cuOccupancyMaxActiveBlocksPerMultiprocessorWithFlags
507
cuOccupancyMaxActiveClusters
508
cuOccupancyMaxPotentialBlockSize
509
cuOccupancyMaxPotentialBlockSizeWithFlags
511
cuOccupancyMaxPotentialClusterSize
512
6.26. Texture Reference Management [DEPRECATED]
513
cuTexRefCreate
513
cuTexRefDestroy
514
cuTexRefGetAddress
514
cuTexRefGetAddressMode
515
cuTexRefGetArray
516
cuTexRefGetBorderColor
516
cuTexRefGetFilterMode
517
cuTexRefGetFlags
518
cuTexRefGetFormat
518
cuTexRefGetMaxAnisotropy
519
cuTexRefGetMipmapFilterMode
520
|
xix
cuTexRefGetMipmapLevelBias
520
cuTexRefGetMipmapLevelClamp
521
cuTexRefGetMipmappedArray
522
cuTexRefSetAddress
522
cuTexRefSetAddress2D
523
cuTexRefSetAddressMode
525
cuTexRefSetArray
526
cuTexRefSetBorderColor
526
cuTexRefSetFilterMode
527
cuTexRefSetFlags
528
cuTexRefSetFormat
529
cuTexRefSetMaxAnisotropy
529
cuTexRefSetMipmapFilterMode
530
cuTexRefSetMipmapLevelBias
531
cuTexRefSetMipmapLevelClamp
531
cuTexRefSetMipmappedArray
532
6.27. Surface Reference Management [DEPRECATED]
533
cuSurfRefGetArray
533
cuSurfRefSetArray
534
6.28. Texture Object Management
534
cuTexObjectCreate
535
cuTexObjectDestroy
539
cuTexObjectGetResourceDesc
540
cuTexObjectGetResourceViewDesc
540
cuTexObjectGetTextureDesc
541
6.29. Surface Object Management
541
cuSurfObjectCreate
542
cuSurfObjectDestroy
542
cuSurfObjectGetResourceDesc
543
6.30. Tensor Map Object Managment
543
cuTensorMapEncodeIm2col
544
cuTensorMapEncodeIm2colWide
549
cuTensorMapEncodeTiled
554
cuTensorMapReplaceAddress
558
6.31. Peer Context Memory Access
559
cuCtxDisablePeerAccess
559
cuCtxEnablePeerAccess
560
cuDeviceCanAccessPeer
561
|
xx
cuDeviceGetP2PAttribute
562
6.32. Graphics Interoperability
563
cuGraphicsMapResources
563
cuGraphicsResourceGetMappedMipmappedArray
564
cuGraphicsResourceGetMappedPointer
565
cuGraphicsResourceSetMapFlags
566
cuGraphicsSubResourceGetMappedArray
567
cuGraphicsUnmapResources
568
cuGraphicsUnregisterResource
569
6.33. Driver Entry Point Access
569
cuGetProcAddress
570
6.34. Coredump Attributes Control API
571
CUCoredumpGenerationFlags
571
CUcoredumpSettings
572
cuCoredumpGetAttribute
572
cuCoredumpGetAttributeGlobal
574
cuCoredumpSetAttribute
575
cuCoredumpSetAttributeGlobal
577
6.35. Green Contexts
579
CUdevResource
580
CUdevSmResource
580
CUdevResourceType
580
CUdevResourceDesc
580
cuCtxFromGreenCtx
580
cuCtxGetDevResource
581
cuDeviceGetDevResource
582
cuDevResourceGenerateDesc
582
cuDevSmResourceSplitByCount
583
cuGreenCtxCreate
585
cuGreenCtxDestroy
586
cuGreenCtxGetDevResource
587
cuGreenCtxRecordEvent
587
cuGreenCtxStreamCreate
588
cuGreenCtxWaitEvent
589
cuStreamGetGreenCtx
590
6.36. Error Log Management Functions
591
cuLogsCurrent
591
cuLogsDumpToFile
592
|
xxi
cuLogsDumpToMemory
592
cuLogsRegisterCallback
593
cuLogsUnregisterCallback
594
6.37. CUDA Checkpointing
594
cuCheckpointProcessCheckpoint
594
cuCheckpointProcessGetRestoreThreadId
595
cuCheckpointProcessGetState
595
cuCheckpointProcessLock
596
cuCheckpointProcessRestore
596
cuCheckpointProcessUnlock
597
6.38. Profiler Control [DEPRECATED]
598
cuProfilerInitialize
598
6.39. Profiler Control
599
cuProfilerStart
599
cuProfilerStop
599
6.40. OpenGL Interoperability
600
OpenGL Interoperability [DEPRECATED]
600
CUGLDeviceList
600
cuGLGetDevices
601
cuGraphicsGLRegisterBuffer
602
cuGraphicsGLRegisterImage
603
cuWGLGetDevice
604
6.40.1. OpenGL Interoperability [DEPRECATED]
605
CUGLmap_flags
605
cuGLCtxCreate
605
cuGLInit
606
cuGLMapBufferObject
607
cuGLMapBufferObjectAsync
608
cuGLRegisterBufferObject
609
cuGLSetBufferObjectMapFlags
609
cuGLUnmapBufferObject
610
cuGLUnmapBufferObjectAsync
611
cuGLUnregisterBufferObject
612
6.41. Direct3D 9 Interoperability
612
Direct3D 9 Interoperability [DEPRECATED]
613
CUd3d9DeviceList
613
cuD3D9CtxCreate
613
cuD3D9CtxCreateOnDevice
614
|
xxii
cuD3D9GetDevice
615
cuD3D9GetDevices
616
cuD3D9GetDirect3DDevice
617
cuGraphicsD3D9RegisterResource
618
6.41.1. Direct3D 9 Interoperability [DEPRECATED]
620
CUd3d9map_flags
620
CUd3d9register_flags
620
cuD3D9MapResources
620
cuD3D9RegisterResource
621
cuD3D9ResourceGetMappedArray
623
cuD3D9ResourceGetMappedPitch
624
cuD3D9ResourceGetMappedPointer
625
cuD3D9ResourceGetMappedSize
626
cuD3D9ResourceGetSurfaceDimensions
627
cuD3D9ResourceSetMapFlags
629
cuD3D9UnmapResources
630
cuD3D9UnregisterResource
631
6.42. Direct3D 10 Interoperability
631
Direct3D 10 Interoperability [DEPRECATED]
631
CUd3d10DeviceList
631
cuD3D10GetDevice
632
cuD3D10GetDevices
633
cuGraphicsD3D10RegisterResource
634
6.42.1. Direct3D 10 Interoperability [DEPRECATED]
636
CUD3D10map_flags
636
CUD3D10register_flags
636
cuD3D10CtxCreate
636
cuD3D10CtxCreateOnDevice
637
cuD3D10GetDirect3DDevice
638
cuD3D10MapResources
639
cuD3D10RegisterResource
640
cuD3D10ResourceGetMappedArray
641
cuD3D10ResourceGetMappedPitch
642
cuD3D10ResourceGetMappedPointer
643
cuD3D10ResourceGetMappedSize
644
cuD3D10ResourceGetSurfaceDimensions
645
cuD3D10ResourceSetMapFlags
646
cuD3D10UnmapResources
647
|
xxiii
cuD3D10UnregisterResource
648
6.43. Direct3D 11 Interoperability
649
Direct3D 11 Interoperability [DEPRECATED]
649
CUd3d11DeviceList
649
cuD3D11GetDevice
650
cuD3D11GetDevices
650
cuGraphicsD3D11RegisterResource
652
6.43.1. Direct3D 11 Interoperability [DEPRECATED]
654
cuD3D11CtxCreate
654
cuD3D11CtxCreateOnDevice
655
cuD3D11GetDirect3DDevice
656
6.44. VDPAU Interoperability
656
cuGraphicsVDPAURegisterOutputSurface
656
cuGraphicsVDPAURegisterVideoSurface
658
cuVDPAUCtxCreate
659
cuVDPAUGetDevice
660
6.45. EGL Interoperability
660
cuEGLStreamConsumerAcquireFrame
661
cuEGLStreamConsumerConnect
662
cuEGLStreamConsumerConnectWithFlags
662
cuEGLStreamConsumerDisconnect
663
cuEGLStreamConsumerReleaseFrame
664
cuEGLStreamProducerConnect
664
cuEGLStreamProducerDisconnect
665
cuEGLStreamProducerPresentFrame
666
cuEGLStreamProducerReturnFrame
667
cuEventCreateFromEGLSync
667
cuGraphicsEGLRegisterImage
668
cuGraphicsResourceGetMappedEglFrame
669
Chapter 7. Data Structures
671
CUaccessPolicyWindow_v1
673
base_ptr
673
hitProp
673
hitRatio
673
missProp
673
num_bytes
673
CUarrayMapInfo_v1
673
deviceBitMask
674
|
xxiv
extentDepth
674
extentHeight
674
extentWidth
674
flags
674
layer
674
level
674
memHandleType
674
memOperationType
674
offset
674
offsetX
675
offsetY
675
offsetZ
675
reserved
675
resourceType
675
size
675
subresourceType
675
CUasyncNotificationInfo
675
bytesOverBudget
675
info
676
overBudget
676
type
676
CUcheckpointCheckpointArgs
676
reserved
676
CUcheckpointLockArgs
676
reserved0
676
reserved1
676
timeoutMs
676
CUcheckpointRestoreArgs
677
reserved
677
CUcheckpointUnlockArgs
677
reserved
677
CUctxCigParam
677
CUctxCreateParams
677
CUDA_ARRAY3D_DESCRIPTOR_v2
677
Depth
677
Flags
678
Format
678
Height
678
|
xxv
NumChannels
678
Width
678
CUDA_ARRAY_DESCRIPTOR_v2
678
Format
678
Height
678
NumChannels
679
Width
679
CUDA_ARRAY_MEMORY_REQUIREMENTS_v1
679
alignment
679
size
679
CUDA_ARRAY_SPARSE_PROPERTIES_v1
679
depth
679
flags
680
height
680
miptailFirstLevel
680
miptailSize
680
width
680
CUDA_CHILD_GRAPH_NODE_PARAMS
680
graph
680
ownership
681
CUDA_CONDITIONAL_NODE_PARAMS
681
ctx
681
handle
681
phGraph_out
681
size
682
type
682
CUDA_EVENT_RECORD_NODE_PARAMS
682
event
682
CUDA_EVENT_WAIT_NODE_PARAMS
682
event
682
CUDA_EXT_SEM_SIGNAL_NODE_PARAMS_v1
683
extSemArray
683
numExtSems
683
paramsArray
683
CUDA_EXT_SEM_SIGNAL_NODE_PARAMS_v2
683
extSemArray
683
numExtSems
683
paramsArray
684
|
xxvi
CUDA_EXT_SEM_WAIT_NODE_PARAMS_v1
684
extSemArray
684
numExtSems
684
paramsArray
684
CUDA_EXT_SEM_WAIT_NODE_PARAMS_v2
684
extSemArray
685
numExtSems
685
paramsArray
685
CUDA_EXTERNAL_MEMORY_BUFFER_DESC_v1
685
flags
685
offset
685
size
685
CUDA_EXTERNAL_MEMORY_HANDLE_DESC_v1
686
fd
686
flags
686
handle
686
name
686
nvSciBufObject
686
size
686
type
687
win32
687
CUDA_EXTERNAL_MEMORY_MIPMAPPED_ARRAY_DESC_v1
687
arrayDesc
687
numLevels
687
offset
688
CUDA_EXTERNAL_SEMAPHORE_HANDLE_DESC_v1
688
fd
688
flags
688
handle
688
name
688
nvSciSyncObj
688
type
689
win32
689
CUDA_EXTERNAL_SEMAPHORE_SIGNAL_PARAMS_v1
689
fence
689
fence
689
flags
690
key
690
|
xxvii
keyedMutex
690
value
690
CUDA_EXTERNAL_SEMAPHORE_WAIT_PARAMS_v1
690
fence
690
flags
691
key
691
keyedMutex
691
nvSciSync
691
timeoutMs
691
value
691
CUDA_GRAPH_INSTANTIATE_PARAMS
692
flags
692
hErrNode_out
692
hUploadStream
692
result_out
692
CUDA_HOST_NODE_PARAMS_v1
692
fn
692
userData
692
CUDA_HOST_NODE_PARAMS_v2
693
fn
693
userData
693
CUDA_KERNEL_NODE_PARAMS_v1
693
blockDimX
693
blockDimY
693
blockDimZ
693
extra
693
func
694
gridDimX
694
gridDimY
694
gridDimZ
694
kernelParams
694
sharedMemBytes
694
CUDA_KERNEL_NODE_PARAMS_v2
694
blockDimX
694
blockDimY
695
blockDimZ
695
ctx
695
extra
695
|
xxviii
func
695
gridDimX
695
gridDimY
695
gridDimZ
695
kern
696
kernelParams
696
sharedMemBytes
696
CUDA_KERNEL_NODE_PARAMS_v3
696
blockDimX
696
blockDimY
696
blockDimZ
696
ctx
696
extra
697
func
697
gridDimX
697
gridDimY
697
gridDimZ
697
kern
697
kernelParams
697
sharedMemBytes
697
CUDA_LAUNCH_PARAMS_v1
698
blockDimX
698
blockDimY
698
blockDimZ
698
function
698
gridDimX
698
gridDimY
698
gridDimZ
698
hStream
698
kernelParams
698
sharedMemBytes
699
CUDA_MEM_ALLOC_NODE_PARAMS_v1
699
accessDescCount
699
accessDescs
699
bytesize
699
dptr
699
poolProps
699
CUDA_MEM_ALLOC_NODE_PARAMS_v2
700
|
xxix
accessDescCount
700
accessDescs
700
bytesize
700
dptr
700
poolProps
700
CUDA_MEM_FREE_NODE_PARAMS
700
dptr
701
CUDA_MEMCPY2D_v2
701
dstArray
701
dstDevice
701
dstHost
701
dstMemoryType
701
dstPitch
701
dstXInBytes
701
dstY
701
Height
702
srcArray
702
srcDevice
702
srcHost
702
srcMemoryType
702
srcPitch
702
srcXInBytes
702
srcY
702
WidthInBytes
702
CUDA_MEMCPY3D_PEER_v1
702
Depth
703
dstArray
703
dstContext
703
dstDevice
703
dstHeight
703
dstHost
703
dstLOD
703
dstMemoryType
703
dstPitch
703
dstXInBytes
703
dstY
704
dstZ
704
Height
704
|
xxx
srcArray
704
srcContext
704
srcDevice
704
srcHeight
704
srcHost
704
srcLOD
704
srcMemoryType
704
srcPitch
705
srcXInBytes
705
srcY
705
srcZ
705
WidthInBytes
705
CUDA_MEMCPY3D_v2
705
Depth
705
dstArray
705
dstDevice
705
dstHeight
705
dstHost
706
dstLOD
706
dstMemoryType
706
dstPitch
706
dstXInBytes
706
dstY
706
dstZ
706
Height
706
reserved0
706
reserved1
706
srcArray
707
srcDevice
707
srcHeight
707
srcHost
707
srcLOD
707
srcMemoryType
707
srcPitch
707
srcXInBytes
707
srcY
707
srcZ
707
WidthInBytes
708
|
xxxi
CUDA_MEMCPY_NODE_PARAMS
708
copyCtx
708
copyParams
708
flags
708
reserved
708
CUDA_MEMSET_NODE_PARAMS_v1
708
dst
708
elementSize
709
height
709
pitch
709
value
709
width
709
CUDA_MEMSET_NODE_PARAMS_v2
709
ctx
709
dst
709
elementSize
709
height
710
pitch
710
value
710
width
710
CUDA_POINTER_ATTRIBUTE_P2P_TOKENS_v1
710
CUDA_RESOURCE_DESC_v1
710
devPtr
710
flags
710
format
710
hArray
711
height
711
hMipmappedArray
711
numChannels
711
pitchInBytes
711
resType
711
sizeInBytes
711
width
711
CUDA_RESOURCE_VIEW_DESC_v1
711
depth
712
firstLayer
712
firstMipmapLevel
712
format
712
|
xxxii
height
712
lastLayer
712
lastMipmapLevel
712
width
712
CUDA_TEXTURE_DESC_v1
713
addressMode
713
borderColor
713
filterMode
713
flags
713
maxAnisotropy
713
maxMipmapLevelClamp
713
minMipmapLevelClamp
713
mipmapFilterMode
714
mipmapLevelBias
714
CUdevprop_v1
714
clockRate
714
maxGridSize
714
maxThreadsDim
714
maxThreadsPerBlock
714
memPitch
714
regsPerBlock
714
sharedMemPerBlock
714
SIMDWidth
715
textureAlign
715
totalConstantMemory
715
CUdevResource
715
CUdevSmResource
715
smCount
715
CUeglFrame_v1
715
cuFormat
716
depth
716
eglColorFormat
716
frameType
716
height
716
numChannels
716
pArray
716
pitch
716
planeCount
716
|
xxxiii
pPitch
716
width
717
CUexecAffinityParam_v1
717
CUexecAffinitySmCount_v1
717
val
717
CUextent3D_v1
717
CUgraphEdgeData
717
from_port
717
reserved
718
to_port
718
type
718
CUgraphExecUpdateResultInfo_v1
718
errorFromNode
718
errorNode
718
result
718
CUgraphNodeParams
719
alloc
719
conditional
719
eventRecord
719
eventWait
719
extSemSignal
719
extSemWait
719
free
719
graph
720
host
720
kernel
720
memcpy
720
memOp
720
memset
720
reserved0
720
reserved1
720
reserved2
721
type
721
CUipcEventHandle_v1
721
CUipcMemHandle_v1
721
CUlaunchAttribute
721
id
721
value
721
|
xxxiv
CUlaunchAttributeValue
721
accessPolicyWindow
721
clusterDim
722
clusterSchedulingPolicyPreference
722
cooperative
722
deviceUpdatableKernelNode
722
launchCompletionEvent
722
memSyncDomain
723
memSyncDomainMap
723
preferredClusterDim
723
priority
723
programmaticEvent
723
programmaticStreamSerializationAllowed
724
sharedMemCarveout
724
syncPolicy
724
CUlaunchConfig
724
attrs
724
blockDimX
724
blockDimY
724
blockDimZ
724
gridDimX
725
gridDimY
725
gridDimZ
725
hStream
725
numAttrs
725
sharedMemBytes
725
CUlaunchMemSyncDomainMap
725
default_
725
remote
726
CUmemAccessDesc_v1
726
flags
726
location
726
CUmemAllocationProp_v1
726
compressionType
726
location
726
requestedHandleTypes
727
type
727
usage
727
|
xxxv
win32HandleMetaData
727
CUmemcpy3DOperand_v1
727
array
727
layerHeight
727
locHint
728
ptr
728
rowLength
728
CUmemcpyAttributes_v1
728
dstLocHint
728
flags
728
srcAccessOrder
728
srcLocHint
729
CUmemDecompressParams
729
algo
729
dst
729
dstActBytes
729
dstNumBytes
729
src
729
srcNumBytes
729
CUmemFabricHandle_v1
730
CUmemLocation_v1
730
id
730
type
730
CUmemPoolProps_v1
730
allocType
730
handleTypes
730
location
730
maxSize
731
reserved
731
usage
731
win32SecurityAttributes
731
CUmemPoolPtrExportData_v1
731
CUmulticastObjectProp_v1
731
flags
731
handleTypes
731
numDevices
732
size
732
CUoffset3D_v1
732
|
xxxvi
CUstreamBatchMemOpParams_v1
732
CUtensorMap
732
7.14. Difference between the driver and runtime APIs
732
Chapter 8. Data Fields
734
Chapter 9. Deprecated List
750
|
xxxvii
|
xxxviii
Chapter 1.
Difference between the driver
and runtime APIs
The driver and runtime APIs are very similar and can for the most part be used interchangeably.
However, there are some key differences worth noting between the two.
Complexity vs. control
The runtime API eases device code management by providing implicit initialization, context
management, and module management. This leads to simpler code, but it also lacks the level of control
that the driver API has.
In comparison, the driver API offers more fine-grained control, especially over contexts and module
loading. Kernel launches are much more complex to implement, as the execution configuration and
kernel parameters must be specified with explicit function calls. However, unlike the runtime, where
all the kernels are automatically loaded during initialization and stay loaded for as long as the program
runs, with the driver API it is possible to only keep the modules that are currently needed loaded, or
even dynamically reload modules. The driver API is also language-independent as it only deals with
cubin objects.
Context management
Context management can be done through the driver API, but is not exposed in the runtime API.
Instead, the runtime API decides itself which context to use for a thread: if a context has been made
current to the calling thread through the driver API, the runtime will use that, but if there is no such
context, it uses a "primary context." Primary contexts are created as needed, one per device per
process, are reference-counted, and are then destroyed when there are no more references to them.
Within one process, all users of the runtime API will share the primary context, unless a context has
been made current to each thread. The context that the runtime uses, i.e, either the current context
or primary context, can be synchronized with cudaDeviceSynchronize(), and destroyed with
cudaDeviceReset().
Using the runtime API with primary contexts has its tradeoffs, however. It can cause trouble for users
writing plug-ins for larger software packages, for example, because if all plug-ins run in the same
process, they will all share a context but will likely have no way to communicate with each other. So,
if one of them calls cudaDeviceReset() after finishing all its CUDA work, the other plug-ins will
fail because the context they were using was destroyed without their knowledge. To avoid this issue,
|
1
Difference between the driver and runtime APIs
CUDA clients can use the driver API to create and set the current context, and then use the runtime API
to work with it. However, contexts may consume significant resources, such as device memory, extra
host threads, and performance costs of context switching on the device. This runtime-driver context
sharing is important when using the driver API in conjunction with libraries built on the runtime API,
such as cuBLAS or cuFFT.
|
2
Chapter 2.
API synchronization behavior
The API provides memcpy/memset functions in both synchronous and asynchronous forms, the
latter having an "Async" suffix. This is a misnomer as each function may exhibit synchronous or
asynchronous behavior depending on the arguments passed to the function. The synchronous forms of
these APIs issue these copies through the default stream.
Any CUDA API call may block or synchronize for various reasons such as contention for or
unavailability of internal resources. Such behavior is subject to change and undocumented behavior
should not be relied upon.
Memcpy
In the reference documentation, each memcpy function is categorized as synchronous or asynchronous,
corresponding to the definitions below.
Synchronous
1. For transfers from pageable host memory to device memory, a stream sync is performed before the
copy is initiated. The function will return once the pageable buffer has been copied to the staging
memory for DMA transfer to device memory, but the DMA to final destination may not have
completed.
2. For transfers from pinned host memory to device memory, the function is synchronous with respect
to the host.
3. For transfers from device to either pageable or pinned host memory, the function returns only once
the copy has completed.
4. For transfers from device memory to device memory, no host-side synchronization is performed.
5. For transfers from any host memory to any host memory, the function is fully synchronous with
respect to the host.
Asynchronous
1. For transfers between device memory and pageable host memory, the function might be
synchronous with respect to host.
2. For transfers from any host memory to any host memory, the function is fully synchronous with
respect to the host.
|
3
API synchronization behavior
3. If pageable memory must first be staged to pinned memory, the driver may synchronize with the
stream and stage the copy into pinned memory.
4. For all other transfers, the function should be fully asynchronous.
Memset
The cudaMemset functions are asynchronous with respect to the host except when the target memory is
pinned host memory. The Async versions are always asynchronous with respect to the host.
Kernel Launches
Kernel launches are asynchronous with respect to the host. Details of concurrent kernel execution and
data transfers can be found in the CUDA Programmers Guide.
|
4
Chapter 3.
Stream synchronization
behavior
Default stream
The default stream, used when 0 is passed as a cudaStream_t or by APIs that operate on a stream
implicitly, can be configured to have either legacy or per-thread synchronization behavior as described
below.
The behavior can be controlled per compilation unit with the --default-stream
nvcc option. Alternatively, per-thread behavior can be enabled by defining the
CUDA_API_PER_THREAD_DEFAULT_STREAM macro before including any CUDA headers. Either way,
the CUDA_API_PER_THREAD_DEFAULT_STREAM macro will be defined in compilation units using per-
thread synchronization behavior.
Legacy default stream
The legacy default stream is an implicit stream which synchronizes with all other streams in the same
CUcontext except for non-blocking streams, described below. (For applications using the runtime
APIs only, there will be one context per device.) When an action is taken in the legacy stream such as a
kernel launch or cudaStreamWaitEvent(), the legacy stream first waits on all blocking streams, the
action is queued in the legacy stream, and then all blocking streams wait on the legacy stream.
For example, the following code launches a kernel k_1 in stream s, then k_2 in the legacy stream, then
k_3 in stream s:
k_1<<<1, 1, 0, s>>>();
k_2<<<1, 1>>>();
k_3<<<1, 1, 0, s>>>();
The resulting behavior is that k_2 will block on k_1 and k_3 will block on k_2.
Non-blocking streams which do not synchronize with the legacy stream can be created using the
cudaStreamNonBlocking flag with the stream creation APIs.
The legacy default stream can be used explicitly with the CUstream (cudaStream_t) handle
CU_STREAM_LEGACY (cudaStreamLegacy).
|
5
Stream synchronization behavior
Per-thread default stream
The per-thread default stream is an implicit stream local to both the thread and the CUcontext, and
which does not synchronize with other streams (just like explicitly created streams). The per-thread
default stream is not a non-blocking stream and will synchronize with the legacy default stream if both
are used in a program.
The per-thread default stream can be used explicitly with the CUstream (cudaStream_t) handle
CU_STREAM_PER_THREAD (cudaStreamPerThread).
|
6
Chapter 4.
Graph object thread safety
Graph objects (cudaGraph_t, CUgraph) are not internally synchronized and must not be accessed
concurrently from multiple threads. API calls accessing the same graph object must be serialized
externally.
Note that this includes APIs which may appear to be read-only, such as cudaGraphClone()
(cuGraphClone()) and cudaGraphInstantiate() (cuGraphInstantiate()). No API or pair
of APIs is guaranteed to be safe to call on the same graph object from two different threads without
serialization.
|
7
Chapter 5.
Rules for version mixing
1. Starting with CUDA 11.0, the ABI version for the CUDA runtime is bumped every major release.
CUDA-defined types, whether opaque handles or structures like cudaDeviceProp, have their
ABI tied to the major release of the CUDA runtime. It is unsafe to pass them from function A to
function B if those functions have been compiled with different major versions of the toolkit and
linked together into the same device executable.
2. The CUDA Driver API has a per-function ABI denoted with a _v* extension. CUDA-defined types
(e.g structs) should not be passed across different ABI versions. For example, an application calling
cuMemcpy2D_v2(const CUDA_MEMCPY2D_v2 *pCopy) and using the older version of the
struct CUDA_MEMCPY2D_v1 instead of CUDA_MEMCPY2D_v2.
3. Users should not arbitrarily mix different API versions during the lifetime of a resource. These
resources include IPC handles, memory, streams, contexts, events, etc. For example, a user
who wants to allocate CUDA memory using cuMemAlloc_v2 should free the memory using
cuMemFree_v2 and not cuMemFree.
|
8
Chapter 6.
Modules
Here is a list of all modules:
‣ Data types used by CUDA driver
‣ Error Handling
‣ Initialization
‣ Version Management
‣ Device Management
‣ Device Management [DEPRECATED]
‣ Primary Context Management
‣ Context Management
‣ Context Management [DEPRECATED]
‣ Module Management
‣ Module Management [DEPRECATED]
‣ Library Management
‣ Memory Management
‣ Virtual Memory Management
‣ Stream Ordered Memory Allocator
‣ Multicast Object Management
‣ Unified Addressing
‣ Stream Management
‣ Event Management
‣ External Resource Interoperability
‣ Stream Memory Operations
‣ Execution Control
‣ Execution Control [DEPRECATED]
‣ Graph Management
‣ Occupancy
‣ Texture Reference Management [DEPRECATED]
|
9
Modules
‣ Surface Reference Management [DEPRECATED]
‣ Texture Object Management
‣ Surface Object Management
‣ Tensor Map Object Managment
‣ Peer Context Memory Access
‣ Graphics Interoperability
‣ Driver Entry Point Access
‣ Coredump Attributes Control API
‣ Green Contexts
‣ Error Log Management Functions
‣ CUDA Checkpointing
‣ Profiler Control [DEPRECATED]
‣ Profiler Control
‣ OpenGL Interoperability
‣ OpenGL Interoperability [DEPRECATED]
‣ Direct3D 9 Interoperability
‣ Direct3D 9 Interoperability [DEPRECATED]
‣ Direct3D 10 Interoperability
‣ Direct3D 10 Interoperability [DEPRECATED]
‣ Direct3D 11 Interoperability
‣ Direct3D 11 Interoperability [DEPRECATED]
‣ VDPAU Interoperability
‣ EGL Interoperability
6.1.
Data types used by CUDA driver
|
10
Modules
struct CUaccessPolicyWindow_v1
struct CUarrayMapInfo_v1
struct CUasyncNotificationInfo
struct CUcheckpointCheckpointArgs
struct CUcheckpointLockArgs
struct CUcheckpointRestoreArgs
struct CUcheckpointUnlockArgs
struct CUctxCigParam
struct CUctxCreateParams
struct CUDA_ARRAY3D_DESCRIPTOR_v2
struct CUDA_ARRAY_DESCRIPTOR_v2
struct CUDA_ARRAY_MEMORY_REQUIREMENTS_v1
struct CUDA_ARRAY_SPARSE_PROPERTIES_v1
struct CUDA_BATCH_MEM_OP_NODE_PARAMS_v2
struct CUDA_CHILD_GRAPH_NODE_PARAMS
struct CUDA_CONDITIONAL_NODE_PARAMS
struct CUDA_EVENT_RECORD_NODE_PARAMS
|
11
Modules
struct CUDA_EVENT_WAIT_NODE_PARAMS
struct CUDA_EXT_SEM_SIGNAL_NODE_PARAMS_v1
struct CUDA_EXT_SEM_SIGNAL_NODE_PARAMS_v2
struct CUDA_EXT_SEM_WAIT_NODE_PARAMS_v1
struct CUDA_EXT_SEM_WAIT_NODE_PARAMS_v2
struct
CUDA_EXTERNAL_MEMORY_BUFFER_DESC_v1
struct
CUDA_EXTERNAL_MEMORY_HANDLE_DESC_v1
struct
CUDA_EXTERNAL_MEMORY_MIPMAPPED_ARRAY_DESC_v
struct
CUDA_EXTERNAL_SEMAPHORE_HANDLE_DESC_v1
struct
CUDA_EXTERNAL_SEMAPHORE_SIGNAL_PARAMS_v1
struct
CUDA_EXTERNAL_SEMAPHORE_WAIT_PARAMS_v1
struct CUDA_GRAPH_INSTANTIATE_PARAMS
struct CUDA_HOST_NODE_PARAMS_v1
|
12
Modules
struct CUDA_HOST_NODE_PARAMS_v2
struct CUDA_KERNEL_NODE_PARAMS_v1
struct CUDA_KERNEL_NODE_PARAMS_v2
struct CUDA_KERNEL_NODE_PARAMS_v3
struct CUDA_LAUNCH_PARAMS_v1
struct CUDA_MEM_ALLOC_NODE_PARAMS_v1
struct CUDA_MEM_ALLOC_NODE_PARAMS_v2
struct CUDA_MEM_FREE_NODE_PARAMS
struct CUDA_MEMCPY2D_v2
struct CUDA_MEMCPY3D_PEER_v1
struct CUDA_MEMCPY3D_v2
struct CUDA_MEMCPY_NODE_PARAMS
struct CUDA_MEMSET_NODE_PARAMS_v1
struct CUDA_MEMSET_NODE_PARAMS_v2
struct
CUDA_POINTER_ATTRIBUTE_P2P_TOKENS_v1
struct CUDA_RESOURCE_DESC_v1
|
13
Modules
struct CUDA_RESOURCE_VIEW_DESC_v1
struct CUDA_TEXTURE_DESC_v1
struct CUdevprop_v1
struct CUeglFrame_v1
struct CUexecAffinityParam_v1
struct CUexecAffinitySmCount_v1
struct CUextent3D_v1
struct CUgraphEdgeData
struct CUgraphExecUpdateResultInfo_v1
struct CUgraphNodeParams
struct CUipcEventHandle_v1
struct CUipcMemHandle_v1
struct CUlaunchAttribute
union CUlaunchAttributeValue
struct CUlaunchConfig
struct CUlaunchMemSyncDomainMap
struct CUmemAccessDesc_v1
|
14
Modules
struct CUmemAllocationProp_v1
struct CUmemcpy3DOperand_v1
struct CUmemcpyAttributes_v1
struct CUmemFabricHandle_v1
struct CUmemLocation_v1
struct CUmemPoolProps_v1
struct CUmemPoolPtrExportData_v1
struct CUmulticastObjectProp_v1
struct CUoffset3D_v1
union CUstreamBatchMemOpParams_v1
struct CUtensorMap
enum cl_context_flags
NVCL context scheduling flags
Values
NVCL_CTX_SCHED_AUTO = 0x00
Automatic scheduling
NVCL_CTX_SCHED_SPIN = 0x01
Set spin as default scheduling
NVCL_CTX_SCHED_YIELD = 0x02
Set yield as default scheduling
NVCL_CTX_SCHED_BLOCKING_SYNC = 0x04
Set blocking synchronization as default scheduling
|
15
Modules
enum cl_event_flags
NVCL event scheduling flags
Values
NVCL_EVENT_SCHED_AUTO = 0x00
Automatic scheduling
NVCL_EVENT_SCHED_SPIN = 0x01
Set spin as default scheduling
NVCL_EVENT_SCHED_YIELD = 0x02
Set yield as default scheduling
NVCL_EVENT_SCHED_BLOCKING_SYNC = 0x04
Set blocking synchronization as default scheduling
enum CUaccessProperty
Specifies performance hint with CUaccessPolicyWindow for hitProp and missProp members.
Values
CU_ACCESS_PROPERTY_NORMAL = 0
Normal cache persistence.
CU_ACCESS_PROPERTY_STREAMING = 1
Streaming access is less likely to persit from cache.
CU_ACCESS_PROPERTY_PERSISTING = 2
Persisting access is more likely to persist in cache.
enum CUaddress_mode
Texture reference addressing modes
Values
CU_TR_ADDRESS_MODE_WRAP = 0
Wrapping address mode
CU_TR_ADDRESS_MODE_CLAMP = 1
Clamp to edge address mode
CU_TR_ADDRESS_MODE_MIRROR = 2
Mirror address mode
CU_TR_ADDRESS_MODE_BORDER = 3
Border address mode
|
16
Modules
enum CUarray_cubemap_face
Array indices for cube faces
Values
CU_CUBEMAP_FACE_POSITIVE_X = 0x00
Positive X face of cubemap
CU_CUBEMAP_FACE_NEGATIVE_X = 0x01
Negative X face of cubemap
CU_CUBEMAP_FACE_POSITIVE_Y = 0x02
Positive Y face of cubemap
CU_CUBEMAP_FACE_NEGATIVE_Y = 0x03
Negative Y face of cubemap
CU_CUBEMAP_FACE_POSITIVE_Z = 0x04
Positive Z face of cubemap
CU_CUBEMAP_FACE_NEGATIVE_Z = 0x05
Negative Z face of cubemap
enum CUarray_format
Array formats
Values
CU_AD_FORMAT_UNSIGNED_INT8 = 0x01
Unsigned 8-bit integers
CU_AD_FORMAT_UNSIGNED_INT16 = 0x02
Unsigned 16-bit integers
CU_AD_FORMAT_UNSIGNED_INT32 = 0x03
Unsigned 32-bit integers
CU_AD_FORMAT_SIGNED_INT8 = 0x08
Signed 8-bit integers
CU_AD_FORMAT_SIGNED_INT16 = 0x09
Signed 16-bit integers
CU_AD_FORMAT_SIGNED_INT32 = 0x0a
Signed 32-bit integers
CU_AD_FORMAT_HALF = 0x10
16-bit floating point
CU_AD_FORMAT_FLOAT = 0x20
32-bit floating point
CU_AD_FORMAT_NV12 = 0xb0
8-bit YUV planar format, with 4:2:0 sampling
CU_AD_FORMAT_UNORM_INT8X1 = 0xc0
|
17
Modules
1 channel unsigned 8-bit normalized integer
CU_AD_FORMAT_UNORM_INT8X2 = 0xc1
2 channel unsigned 8-bit normalized integer
CU_AD_FORMAT_UNORM_INT8X4 = 0xc2
4 channel unsigned 8-bit normalized integer
CU_AD_FORMAT_UNORM_INT16X1 = 0xc3
1 channel unsigned 16-bit normalized integer
CU_AD_FORMAT_UNORM_INT16X2 = 0xc4
2 channel unsigned 16-bit normalized integer
CU_AD_FORMAT_UNORM_INT16X4 = 0xc5
4 channel unsigned 16-bit normalized integer
CU_AD_FORMAT_SNORM_INT8X1 = 0xc6
1 channel signed 8-bit normalized integer
CU_AD_FORMAT_SNORM_INT8X2 = 0xc7
2 channel signed 8-bit normalized integer
CU_AD_FORMAT_SNORM_INT8X4 = 0xc8
4 channel signed 8-bit normalized integer
CU_AD_FORMAT_SNORM_INT16X1 = 0xc9
1 channel signed 16-bit normalized integer
CU_AD_FORMAT_SNORM_INT16X2 = 0xca
2 channel signed 16-bit normalized integer
CU_AD_FORMAT_SNORM_INT16X4 = 0xcb
4 channel signed 16-bit normalized integer
CU_AD_FORMAT_BC1_UNORM = 0x91
4 channel unsigned normalized block-compressed (BC1 compression) format
CU_AD_FORMAT_BC1_UNORM_SRGB = 0x92
4 channel unsigned normalized block-compressed (BC1 compression) format with sRGB encoding
CU_AD_FORMAT_BC2_UNORM = 0x93
4 channel unsigned normalized block-compressed (BC2 compression) format
CU_AD_FORMAT_BC2_UNORM_SRGB = 0x94
4 channel unsigned normalized block-compressed (BC2 compression) format with sRGB encoding
CU_AD_FORMAT_BC3_UNORM = 0x95
4 channel unsigned normalized block-compressed (BC3 compression) format
CU_AD_FORMAT_BC3_UNORM_SRGB = 0x96
4 channel unsigned normalized block-compressed (BC3 compression) format with sRGB encoding
CU_AD_FORMAT_BC4_UNORM = 0x97
1 channel unsigned normalized block-compressed (BC4 compression) format
CU_AD_FORMAT_BC4_SNORM = 0x98
1 channel signed normalized block-compressed (BC4 compression) format
CU_AD_FORMAT_BC5_UNORM = 0x99
2 channel unsigned normalized block-compressed (BC5 compression) format
CU_AD_FORMAT_BC5_SNORM = 0x9a
2 channel signed normalized block-compressed (BC5 compression) format
|
18
Modules
CU_AD_FORMAT_BC6H_UF16 = 0x9b
3 channel unsigned half-float block-compressed (BC6H compression) format
CU_AD_FORMAT_BC6H_SF16 = 0x9c
3 channel signed half-float block-compressed (BC6H compression) format
CU_AD_FORMAT_BC7_UNORM = 0x9d
4 channel unsigned normalized block-compressed (BC7 compression) format
CU_AD_FORMAT_BC7_UNORM_SRGB = 0x9e
4 channel unsigned normalized block-compressed (BC7 compression) format with sRGB encoding
CU_AD_FORMAT_P010 = 0x9f
10-bit YUV planar format, with 4:2:0 sampling
CU_AD_FORMAT_P016 = 0xa1
16-bit YUV planar format, with 4:2:0 sampling
CU_AD_FORMAT_NV16 = 0xa2
8-bit YUV planar format, with 4:2:2 sampling
CU_AD_FORMAT_P210 = 0xa3
10-bit YUV planar format, with 4:2:2 sampling
CU_AD_FORMAT_P216 = 0xa4
16-bit YUV planar format, with 4:2:2 sampling
CU_AD_FORMAT_YUY2 = 0xa5
2 channel, 8-bit YUV packed planar format, with 4:2:2 sampling
CU_AD_FORMAT_Y210 = 0xa6
2 channel, 10-bit YUV packed planar format, with 4:2:2 sampling
CU_AD_FORMAT_Y216 = 0xa7
2 channel, 16-bit YUV packed planar format, with 4:2:2 sampling
CU_AD_FORMAT_AYUV = 0xa8
4 channel, 8-bit YUV packed planar format, with 4:4:4 sampling
CU_AD_FORMAT_Y410 = 0xa9
10-bit YUV packed planar format, with 4:4:4 sampling
CU_AD_FORMAT_Y416 = 0xb1
4 channel, 12-bit YUV packed planar format, with 4:4:4 sampling
CU_AD_FORMAT_Y444_PLANAR8 = 0xb2
3 channel 8-bit YUV planar format, with 4:4:4 sampling
CU_AD_FORMAT_Y444_PLANAR10 = 0xb3
3 channel 10-bit YUV planar format, with 4:4:4 sampling
CU_AD_FORMAT_YUV444_8bit_SemiPlanar = 0xb4
3 channel 8-bit YUV semi-planar format, with 4:4:4 sampling
CU_AD_FORMAT_YUV444_16bit_SemiPlanar = 0xb5
3 channel 16-bit YUV semi-planar format, with 4:4:4 sampling
CU_AD_FORMAT_UNORM_INT_101010_2 = 0x50
4 channel unorm R10G10B10A2 RGB format
CU_AD_FORMAT_MAX = 0x7FFFFFFF
|
19
Modules
enum CUarraySparseSubresourceType
Sparse subresource types
Values
CU_ARRAY_SPARSE_SUBRESOURCE_TYPE_SPARSE_LEVEL = 0
CU_ARRAY_SPARSE_SUBRESOURCE_TYPE_MIPTAIL = 1
enum CUasyncNotificationType
Types of async notification that can be sent
Values
CU_ASYNC_NOTIFICATION_TYPE_OVER_BUDGET = 0x1
Sent when the process has exceeded its device memory budget
enum CUclusterSchedulingPolicy
Cluster scheduling policies. These may be passed to cuFuncSetAttribute or cuKernelSetAttribute
Values
CU_CLUSTER_SCHEDULING_POLICY_DEFAULT = 0
the default policy
CU_CLUSTER_SCHEDULING_POLICY_SPREAD = 1
spread the blocks within a cluster to the SMs
CU_CLUSTER_SCHEDULING_POLICY_LOAD_BALANCING = 2
allow the hardware to load-balance the blocks in a cluster to the SMs
enum CUcomputemode
Compute Modes
Values
CU_COMPUTEMODE_DEFAULT = 0
Default compute mode (Multiple contexts allowed per device)
CU_COMPUTEMODE_PROHIBITED = 2
Compute-prohibited mode (No contexts can be created on this device at this time)
CU_COMPUTEMODE_EXCLUSIVE_PROCESS = 3
Compute-exclusive-process mode (Only one context used by a single process can be present on this
device at a time)
|
20
Modules
enum CUctx_flags
Context creation flags
Values
CU_CTX_SCHED_AUTO = 0x00
Automatic scheduling
CU_CTX_SCHED_SPIN = 0x01
Set spin as default scheduling
CU_CTX_SCHED_YIELD = 0x02
Set yield as default scheduling
CU_CTX_SCHED_BLOCKING_SYNC = 0x04
Set blocking synchronization as default scheduling
CU_CTX_BLOCKING_SYNC = 0x04
Set blocking synchronization as default scheduling Deprecated This flag was deprecated as of
CUDA 4.0 and was replaced with CU_CTX_SCHED_BLOCKING_SYNC.
CU_CTX_SCHED_MASK = 0x07
CU_CTX_MAP_HOST = 0x08
Deprecated This flag was deprecated as of CUDA 11.0 and it no longer has any effect. All contexts
as of CUDA 3.2 behave as though the flag is enabled.
CU_CTX_LMEM_RESIZE_TO_MAX = 0x10
Keep local memory allocation after launch
CU_CTX_COREDUMP_ENABLE = 0x20
Trigger coredumps from exceptions in this context
CU_CTX_USER_COREDUMP_ENABLE = 0x40
Enable user pipe to trigger coredumps in this context
CU_CTX_SYNC_MEMOPS = 0x80
Ensure synchronous memory operations on this context will synchronize
CU_CTX_FLAGS_MASK = 0xFF
enum CUDA_POINTER_ATTRIBUTE_ACCESS_FLAGS
Access flags that specify the level of access the current context's device has on the memory referenced.
Values
CU_POINTER_ATTRIBUTE_ACCESS_FLAG_NONE = 0x0
No access, meaning the device cannot access this memory at all, thus must be staged through
accessible memory in order to complete certain operations
CU_POINTER_ATTRIBUTE_ACCESS_FLAG_READ = 0x1
Read-only access, meaning writes to this memory are considered invalid accesses and thus return
error in that case.
CU_POINTER_ATTRIBUTE_ACCESS_FLAG_READWRITE = 0x3
|
21
Modules
Read-write access, the device has full read-write access to the memory
enum CUdevice_attribute
Device properties
Values
CU_DEVICE_ATTRIBUTE_MAX_THREADS_PER_BLOCK = 1
Maximum number of threads per block
CU_DEVICE_ATTRIBUTE_MAX_BLOCK_DIM_X = 2
Maximum block dimension X
CU_DEVICE_ATTRIBUTE_MAX_BLOCK_DIM_Y = 3
Maximum block dimension Y
CU_DEVICE_ATTRIBUTE_MAX_BLOCK_DIM_Z = 4
Maximum block dimension Z
CU_DEVICE_ATTRIBUTE_MAX_GRID_DIM_X = 5
Maximum grid dimension X
CU_DEVICE_ATTRIBUTE_MAX_GRID_DIM_Y = 6
Maximum grid dimension Y
CU_DEVICE_ATTRIBUTE_MAX_GRID_DIM_Z = 7
Maximum grid dimension Z
CU_DEVICE_ATTRIBUTE_MAX_SHARED_MEMORY_PER_BLOCK = 8
Maximum shared memory available per block in bytes
CU_DEVICE_ATTRIBUTE_SHARED_MEMORY_PER_BLOCK = 8
Deprecated, use CU_DEVICE_ATTRIBUTE_MAX_SHARED_MEMORY_PER_BLOCK
CU_DEVICE_ATTRIBUTE_TOTAL_CONSTANT_MEMORY = 9
Memory available on device for __constant__ variables in a CUDA C kernel in bytes
CU_DEVICE_ATTRIBUTE_WARP_SIZE = 10
Warp size in threads
CU_DEVICE_ATTRIBUTE_MAX_PITCH = 11
Maximum pitch in bytes allowed by memory copies
CU_DEVICE_ATTRIBUTE_MAX_REGISTERS_PER_BLOCK = 12
Maximum number of 32-bit registers available per block
CU_DEVICE_ATTRIBUTE_REGISTERS_PER_BLOCK = 12
Deprecated, use CU_DEVICE_ATTRIBUTE_MAX_REGISTERS_PER_BLOCK
CU_DEVICE_ATTRIBUTE_CLOCK_RATE = 13
Typical clock frequency in kilohertz
CU_DEVICE_ATTRIBUTE_TEXTURE_ALIGNMENT = 14
Alignment requirement for textures
CU_DEVICE_ATTRIBUTE_GPU_OVERLAP = 15
Device can possibly copy memory and execute a kernel concurrently. Deprecated. Use instead
CU_DEVICE_ATTRIBUTE_ASYNC_ENGINE_COUNT.
CU_DEVICE_ATTRIBUTE_MULTIPROCESSOR_COUNT = 16
|
22
Modules
Number of multiprocessors on device
CU_DEVICE_ATTRIBUTE_KERNEL_EXEC_TIMEOUT = 17
Specifies whether there is a run time limit on kernels
CU_DEVICE_ATTRIBUTE_INTEGRATED = 18
Device is integrated with host memory
CU_DEVICE_ATTRIBUTE_CAN_MAP_HOST_MEMORY = 19
Device can map host memory into CUDA address space
CU_DEVICE_ATTRIBUTE_COMPUTE_MODE = 20
Compute mode (See CUcomputemode for details)
CU_DEVICE_ATTRIBUTE_MAXIMUM_TEXTURE1D_WIDTH = 21
Maximum 1D texture width
CU_DEVICE_ATTRIBUTE_MAXIMUM_TEXTURE2D_WIDTH = 22
Maximum 2D texture width
CU_DEVICE_ATTRIBUTE_MAXIMUM_TEXTURE2D_HEIGHT = 23
Maximum 2D texture height
CU_DEVICE_ATTRIBUTE_MAXIMUM_TEXTURE3D_WIDTH = 24
Maximum 3D texture width
CU_DEVICE_ATTRIBUTE_MAXIMUM_TEXTURE3D_HEIGHT = 25
Maximum 3D texture height
CU_DEVICE_ATTRIBUTE_MAXIMUM_TEXTURE3D_DEPTH = 26
Maximum 3D texture depth
CU_DEVICE_ATTRIBUTE_MAXIMUM_TEXTURE2D_LAYERED_WIDTH = 27
Maximum 2D layered texture width
CU_DEVICE_ATTRIBUTE_MAXIMUM_TEXTURE2D_LAYERED_HEIGHT = 28
Maximum 2D layered texture height
CU_DEVICE_ATTRIBUTE_MAXIMUM_TEXTURE2D_LAYERED_LAYERS = 29
Maximum layers in a 2D layered texture
CU_DEVICE_ATTRIBUTE_MAXIMUM_TEXTURE2D_ARRAY_WIDTH = 27
Deprecated, use CU_DEVICE_ATTRIBUTE_MAXIMUM_TEXTURE2D_LAYERED_WIDTH
CU_DEVICE_ATTRIBUTE_MAXIMUM_TEXTURE2D_ARRAY_HEIGHT = 28
Deprecated, use CU_DEVICE_ATTRIBUTE_MAXIMUM_TEXTURE2D_LAYERED_HEIGHT
CU_DEVICE_ATTRIBUTE_MAXIMUM_TEXTURE2D_ARRAY_NUMSLICES = 29
Deprecated, use CU_DEVICE_ATTRIBUTE_MAXIMUM_TEXTURE2D_LAYERED_LAYERS
CU_DEVICE_ATTRIBUTE_SURFACE_ALIGNMENT = 30
Alignment requirement for surfaces
CU_DEVICE_ATTRIBUTE_CONCURRENT_KERNELS = 31
Device can possibly execute multiple kernels concurrently
CU_DEVICE_ATTRIBUTE_ECC_ENABLED = 32
Device has ECC support enabled
CU_DEVICE_ATTRIBUTE_PCI_BUS_ID = 33
PCI bus ID of the device
CU_DEVICE_ATTRIBUTE_PCI_DEVICE_ID = 34
PCI device ID of the device
|
23
Modules
CU_DEVICE_ATTRIBUTE_TCC_DRIVER = 35
Device is using TCC driver model
CU_DEVICE_ATTRIBUTE_MEMORY_CLOCK_RATE = 36
Peak memory clock frequency in kilohertz
CU_DEVICE_ATTRIBUTE_GLOBAL_MEMORY_BUS_WIDTH = 37
Global memory bus width in bits
CU_DEVICE_ATTRIBUTE_L2_CACHE_SIZE = 38
Size of L2 cache in bytes
CU_DEVICE_ATTRIBUTE_MAX_THREADS_PER_MULTIPROCESSOR = 39
Maximum resident threads per multiprocessor
CU_DEVICE_ATTRIBUTE_ASYNC_ENGINE_COUNT = 40
Number of asynchronous engines
CU_DEVICE_ATTRIBUTE_UNIFIED_ADDRESSING = 41
Device shares a unified address space with the host
CU_DEVICE_ATTRIBUTE_MAXIMUM_TEXTURE1D_LAYERED_WIDTH = 42
Maximum 1D layered texture width
CU_DEVICE_ATTRIBUTE_MAXIMUM_TEXTURE1D_LAYERED_LAYERS = 43
Maximum layers in a 1D layered texture
CU_DEVICE_ATTRIBUTE_CAN_TEX2D_GATHER = 44
Deprecated, do not use.
CU_DEVICE_ATTRIBUTE_MAXIMUM_TEXTURE2D_GATHER_WIDTH = 45
Maximum 2D texture width if CUDA_ARRAY3D_TEXTURE_GATHER is set
CU_DEVICE_ATTRIBUTE_MAXIMUM_TEXTURE2D_GATHER_HEIGHT = 46
Maximum 2D texture height if CUDA_ARRAY3D_TEXTURE_GATHER is set
CU_DEVICE_ATTRIBUTE_MAXIMUM_TEXTURE3D_WIDTH_ALTERNATE = 47
Alternate maximum 3D texture width
CU_DEVICE_ATTRIBUTE_MAXIMUM_TEXTURE3D_HEIGHT_ALTERNATE = 48
Alternate maximum 3D texture height
CU_DEVICE_ATTRIBUTE_MAXIMUM_TEXTURE3D_DEPTH_ALTERNATE = 49
Alternate maximum 3D texture depth
CU_DEVICE_ATTRIBUTE_PCI_DOMAIN_ID = 50
PCI domain ID of the device
CU_DEVICE_ATTRIBUTE_TEXTURE_PITCH_ALIGNMENT = 51
Pitch alignment requirement for textures
CU_DEVICE_ATTRIBUTE_MAXIMUM_TEXTURECUBEMAP_WIDTH = 52
Maximum cubemap texture width/height
CU_DEVICE_ATTRIBUTE_MAXIMUM_TEXTURECUBEMAP_LAYERED_WIDTH = 53
Maximum cubemap layered texture width/height
CU_DEVICE_ATTRIBUTE_MAXIMUM_TEXTURECUBEMAP_LAYERED_LAYERS = 54
Maximum layers in a cubemap layered texture
CU_DEVICE_ATTRIBUTE_MAXIMUM_SURFACE1D_WIDTH = 55
Maximum 1D surface width
CU_DEVICE_ATTRIBUTE_MAXIMUM_SURFACE2D_WIDTH = 56
|
24
Modules
Maximum 2D surface width
CU_DEVICE_ATTRIBUTE_MAXIMUM_SURFACE2D_HEIGHT = 57
Maximum 2D surface height
CU_DEVICE_ATTRIBUTE_MAXIMUM_SURFACE3D_WIDTH = 58
Maximum 3D surface width
CU_DEVICE_ATTRIBUTE_MAXIMUM_SURFACE3D_HEIGHT = 59
Maximum 3D surface height
CU_DEVICE_ATTRIBUTE_MAXIMUM_SURFACE3D_DEPTH = 60
Maximum 3D surface depth
CU_DEVICE_ATTRIBUTE_MAXIMUM_SURFACE1D_LAYERED_WIDTH = 61
Maximum 1D layered surface width
CU_DEVICE_ATTRIBUTE_MAXIMUM_SURFACE1D_LAYERED_LAYERS = 62
Maximum layers in a 1D layered surface
CU_DEVICE_ATTRIBUTE_MAXIMUM_SURFACE2D_LAYERED_WIDTH = 63
Maximum 2D layered surface width
CU_DEVICE_ATTRIBUTE_MAXIMUM_SURFACE2D_LAYERED_HEIGHT = 64
Maximum 2D layered surface height
CU_DEVICE_ATTRIBUTE_MAXIMUM_SURFACE2D_LAYERED_LAYERS = 65
Maximum layers in a 2D layered surface
CU_DEVICE_ATTRIBUTE_MAXIMUM_SURFACECUBEMAP_WIDTH = 66
Maximum cubemap surface width
CU_DEVICE_ATTRIBUTE_MAXIMUM_SURFACECUBEMAP_LAYERED_WIDTH = 67
Maximum cubemap layered surface width
CU_DEVICE_ATTRIBUTE_MAXIMUM_SURFACECUBEMAP_LAYERED_LAYERS = 68
Maximum layers in a cubemap layered surface
CU_DEVICE_ATTRIBUTE_MAXIMUM_TEXTURE1D_LINEAR_WIDTH = 69
Deprecated, do not use. Use cudaDeviceGetTexture1DLinearMaxWidth() or
cuDeviceGetTexture1DLinearMaxWidth() instead.
CU_DEVICE_ATTRIBUTE_MAXIMUM_TEXTURE2D_LINEAR_WIDTH = 70
Maximum 2D linear texture width
CU_DEVICE_ATTRIBUTE_MAXIMUM_TEXTURE2D_LINEAR_HEIGHT = 71
Maximum 2D linear texture height
CU_DEVICE_ATTRIBUTE_MAXIMUM_TEXTURE2D_LINEAR_PITCH = 72
Maximum 2D linear texture pitch in bytes
CU_DEVICE_ATTRIBUTE_MAXIMUM_TEXTURE2D_MIPMAPPED_WIDTH = 73
Maximum mipmapped 2D texture width
CU_DEVICE_ATTRIBUTE_MAXIMUM_TEXTURE2D_MIPMAPPED_HEIGHT = 74
Maximum mipmapped 2D texture height
CU_DEVICE_ATTRIBUTE_COMPUTE_CAPABILITY_MAJOR = 75
Major compute capability version number
CU_DEVICE_ATTRIBUTE_COMPUTE_CAPABILITY_MINOR = 76
Minor compute capability version number
CU_DEVICE_ATTRIBUTE_MAXIMUM_TEXTURE1D_MIPMAPPED_WIDTH = 77
|
25
Modules
Maximum mipmapped 1D texture width
CU_DEVICE_ATTRIBUTE_STREAM_PRIORITIES_SUPPORTED = 78
Device supports stream priorities
CU_DEVICE_ATTRIBUTE_GLOBAL_L1_CACHE_SUPPORTED = 79
Device supports caching globals in L1
CU_DEVICE_ATTRIBUTE_LOCAL_L1_CACHE_SUPPORTED = 80
Device supports caching locals in L1
CU_DEVICE_ATTRIBUTE_MAX_SHARED_MEMORY_PER_MULTIPROCESSOR = 81
Maximum shared memory available per multiprocessor in bytes
CU_DEVICE_ATTRIBUTE_MAX_REGISTERS_PER_MULTIPROCESSOR = 82
Maximum number of 32-bit registers available per multiprocessor
CU_DEVICE_ATTRIBUTE_MANAGED_MEMORY = 83
Device can allocate managed memory on this system
CU_DEVICE_ATTRIBUTE_MULTI_GPU_BOARD = 84
Device is on a multi-GPU board
CU_DEVICE_ATTRIBUTE_MULTI_GPU_BOARD_GROUP_ID = 85
Unique id for a group of devices on the same multi-GPU board
CU_DEVICE_ATTRIBUTE_HOST_NATIVE_ATOMIC_SUPPORTED = 86
Link between the device and the host supports native atomic operations
CU_DEVICE_ATTRIBUTE_SINGLE_TO_DOUBLE_PRECISION_PERF_RATIO = 87
Ratio of single precision performance (in floating-point operations per second) to double precision
performance
CU_DEVICE_ATTRIBUTE_PAGEABLE_MEMORY_ACCESS = 88
Device supports coherently accessing pageable memory without calling cudaHostRegister on it
CU_DEVICE_ATTRIBUTE_CONCURRENT_MANAGED_ACCESS = 89
Device can coherently access managed memory concurrently with the CPU
CU_DEVICE_ATTRIBUTE_COMPUTE_PREEMPTION_SUPPORTED = 90
Device supports compute preemption.
CU_DEVICE_ATTRIBUTE_CAN_USE_HOST_POINTER_FOR_REGISTERED_MEM = 91
Device can access host registered memory at the same virtual address as the CPU
CU_DEVICE_ATTRIBUTE_CAN_USE_STREAM_MEM_OPS_V1 = 92
Deprecated, along with v1 MemOps API, cuStreamBatchMemOp and related APIs are supported.
CU_DEVICE_ATTRIBUTE_CAN_USE_64_BIT_STREAM_MEM_OPS_V1 = 93
Deprecated, along with v1 MemOps API, 64-bit operations are supported in cuStreamBatchMemOp
and related APIs.
CU_DEVICE_ATTRIBUTE_CAN_USE_STREAM_WAIT_VALUE_NOR_V1 = 94
Deprecated, along with v1 MemOps API, CU_STREAM_WAIT_VALUE_NOR is supported.
CU_DEVICE_ATTRIBUTE_COOPERATIVE_LAUNCH = 95
Device supports launching cooperative kernels via cuLaunchCooperativeKernel
CU_DEVICE_ATTRIBUTE_COOPERATIVE_MULTI_DEVICE_LAUNCH = 96
Deprecated, cuLaunchCooperativeKernelMultiDevice is deprecated.
CU_DEVICE_ATTRIBUTE_MAX_SHARED_MEMORY_PER_BLOCK_OPTIN = 97
Maximum optin shared memory per block
|
26
Modules
CU_DEVICE_ATTRIBUTE_CAN_FLUSH_REMOTE_WRITES = 98
The CU_STREAM_WAIT_VALUE_FLUSH flag and the
CU_STREAM_MEM_OP_FLUSH_REMOTE_WRITES MemOp are supported on the device. See
Stream Memory Operations for additional details.
CU_DEVICE_ATTRIBUTE_HOST_REGISTER_SUPPORTED = 99
Device supports host memory registration via cudaHostRegister.
CU_DEVICE_ATTRIBUTE_PAGEABLE_MEMORY_ACCESS_USES_HOST_PAGE_TABLES
= 100
Device accesses pageable memory via the host's page tables.
CU_DEVICE_ATTRIBUTE_DIRECT_MANAGED_MEM_ACCESS_FROM_HOST = 101
The host can directly access managed memory on the device without migration.
CU_DEVICE_ATTRIBUTE_VIRTUAL_ADDRESS_MANAGEMENT_SUPPORTED = 102
Deprecated, Use
CU_DEVICE_ATTRIBUTE_VIRTUAL_MEMORY_MANAGEMENT_SUPPORTED
CU_DEVICE_ATTRIBUTE_VIRTUAL_MEMORY_MANAGEMENT_SUPPORTED = 102
Device supports virtual memory management APIs like cuMemAddressReserve, cuMemCreate,
cuMemMap and related APIs
CU_DEVICE_ATTRIBUTE_HANDLE_TYPE_POSIX_FILE_DESCRIPTOR_SUPPORTED =
103
Device supports exporting memory to a posix file descriptor with
cuMemExportToShareableHandle, if requested via cuMemCreate
CU_DEVICE_ATTRIBUTE_HANDLE_TYPE_WIN32_HANDLE_SUPPORTED = 104
Device supports exporting memory to a Win32 NT handle with cuMemExportToShareableHandle,
if requested via cuMemCreate
CU_DEVICE_ATTRIBUTE_HANDLE_TYPE_WIN32_KMT_HANDLE_SUPPORTED = 105
Device supports exporting memory to a Win32 KMT handle with
cuMemExportToShareableHandle, if requested via cuMemCreate
CU_DEVICE_ATTRIBUTE_MAX_BLOCKS_PER_MULTIPROCESSOR = 106
Maximum number of blocks per multiprocessor
CU_DEVICE_ATTRIBUTE_GENERIC_COMPRESSION_SUPPORTED = 107
Device supports compression of memory
CU_DEVICE_ATTRIBUTE_MAX_PERSISTING_L2_CACHE_SIZE = 108
Maximum L2 persisting lines capacity setting in bytes.
CU_DEVICE_ATTRIBUTE_MAX_ACCESS_POLICY_WINDOW_SIZE = 109
Maximum value of CUaccessPolicyWindow::num_bytes.
CU_DEVICE_ATTRIBUTE_GPU_DIRECT_RDMA_WITH_CUDA_VMM_SUPPORTED =
110
Device supports specifying the GPUDirect RDMA flag with cuMemCreate
CU_DEVICE_ATTRIBUTE_RESERVED_SHARED_MEMORY_PER_BLOCK = 111
Shared memory reserved by CUDA driver per block in bytes
CU_DEVICE_ATTRIBUTE_SPARSE_CUDA_ARRAY_SUPPORTED = 112
Device supports sparse CUDA arrays and sparse CUDA mipmapped arrays
CU_DEVICE_ATTRIBUTE_READ_ONLY_HOST_REGISTER_SUPPORTED = 113
|
27
Modules
Device supports using the cuMemHostRegister flag CU_MEMHOSTERGISTER_READ_ONLY to
register memory that must be mapped as read-only to the GPU
CU_DEVICE_ATTRIBUTE_TIMELINE_SEMAPHORE_INTEROP_SUPPORTED = 114
External timeline semaphore interop is supported on the device
CU_DEVICE_ATTRIBUTE_MEMORY_POOLS_SUPPORTED = 115
Device supports using the cuMemAllocAsync and cuMemPool family of APIs
CU_DEVICE_ATTRIBUTE_GPU_DIRECT_RDMA_SUPPORTED = 116
Device supports GPUDirect RDMA APIs, like nvidia_p2p_get_pages (see https://docs.nvidia.com/
cuda/gpudirect-rdma for more information)
CU_DEVICE_ATTRIBUTE_GPU_DIRECT_RDMA_FLUSH_WRITES_OPTIONS = 117
The returned attribute shall be interpreted as a bitmask, where the individual bits are described by
the CUflushGPUDirectRDMAWritesOptions enum
CU_DEVICE_ATTRIBUTE_GPU_DIRECT_RDMA_WRITES_ORDERING = 118
GPUDirect RDMA writes to the device do not need to be flushed for consumers within the scope
indicated by the returned attribute. See CUGPUDirectRDMAWritesOrdering for the numerical
values returned here.
CU_DEVICE_ATTRIBUTE_MEMPOOL_SUPPORTED_HANDLE_TYPES = 119
Handle types supported with mempool based IPC
CU_DEVICE_ATTRIBUTE_CLUSTER_LAUNCH = 120
Indicates device supports cluster launch
CU_DEVICE_ATTRIBUTE_DEFERRED_MAPPING_CUDA_ARRAY_SUPPORTED = 121
Device supports deferred mapping CUDA arrays and CUDA mipmapped arrays
CU_DEVICE_ATTRIBUTE_CAN_USE_64_BIT_STREAM_MEM_OPS = 122
64-bit operations are supported in cuStreamBatchMemOp and related MemOp APIs.
CU_DEVICE_ATTRIBUTE_CAN_USE_STREAM_WAIT_VALUE_NOR = 123
CU_STREAM_WAIT_VALUE_NOR is supported by MemOp APIs.
CU_DEVICE_ATTRIBUTE_DMA_BUF_SUPPORTED = 124
Device supports buffer sharing with dma_buf mechanism.
CU_DEVICE_ATTRIBUTE_IPC_EVENT_SUPPORTED = 125
Device supports IPC Events.
CU_DEVICE_ATTRIBUTE_MEM_SYNC_DOMAIN_COUNT = 126
Number of memory domains the device supports.
CU_DEVICE_ATTRIBUTE_TENSOR_MAP_ACCESS_SUPPORTED = 127
Device supports accessing memory using Tensor Map.
CU_DEVICE_ATTRIBUTE_HANDLE_TYPE_FABRIC_SUPPORTED = 128
Device supports exporting memory to a fabric handle with cuMemExportToShareableHandle() or
requested with cuMemCreate()
CU_DEVICE_ATTRIBUTE_UNIFIED_FUNCTION_POINTERS = 129
Device supports unified function pointers.
CU_DEVICE_ATTRIBUTE_NUMA_CONFIG = 130
NUMA configuration of a device: value is of type CUdeviceNumaConfig enum
CU_DEVICE_ATTRIBUTE_NUMA_ID = 131
NUMA node ID of the GPU memory
|
28
Modules
CU_DEVICE_ATTRIBUTE_MULTICAST_SUPPORTED = 132
Device supports switch multicast and reduction operations.
CU_DEVICE_ATTRIBUTE_MPS_ENABLED = 133
Indicates if contexts created on this device will be shared via MPS
CU_DEVICE_ATTRIBUTE_HOST_NUMA_ID = 134
NUMA ID of the host node closest to the device. Returns -1 when system does not support NUMA.
CU_DEVICE_ATTRIBUTE_D3D12_CIG_SUPPORTED = 135
Device supports CIG with D3D12.
CU_DEVICE_ATTRIBUTE_MEM_DECOMPRESS_ALGORITHM_MASK = 136
The returned valued shall be interpreted as a bitmask, where the individual bits are described by the
CUmemDecompressAlgorithm enum.
CU_DEVICE_ATTRIBUTE_MEM_DECOMPRESS_MAXIMUM_LENGTH = 137
The returned valued is the maximum length in bytes of a single decompress operation that is
allowed.
CU_DEVICE_ATTRIBUTE_VULKAN_CIG_SUPPORTED = 138
Device supports CIG with Vulkan.
CU_DEVICE_ATTRIBUTE_GPU_PCI_DEVICE_ID = 139
The combined 16-bit PCI device ID and 16-bit PCI vendor ID.
CU_DEVICE_ATTRIBUTE_GPU_PCI_SUBSYSTEM_ID = 140
The combined 16-bit PCI subsystem ID and 16-bit PCI subsystem vendor ID.
CU_DEVICE_ATTRIBUTE_HOST_NUMA_VIRTUAL_MEMORY_MANAGEMENT_SUPPORTED
= 141
Device supports HOST_NUMA location with the virtual memory management APIs like
cuMemCreate, cuMemMap and related APIs
CU_DEVICE_ATTRIBUTE_HOST_NUMA_MEMORY_POOLS_SUPPORTED = 142
Device supports HOST_NUMA location with the cuMemAllocAsync and cuMemPool family of
APIs
CU_DEVICE_ATTRIBUTE_HOST_NUMA_MULTINODE_IPC_SUPPORTED = 143
Device supports HOST_NUMA location IPC between nodes in a multi-node system.
CU_DEVICE_ATTRIBUTE_MAX
enum CUdevice_P2PAttribute
P2P Attributes
Values
CU_DEVICE_P2P_ATTRIBUTE_PERFORMANCE_RANK = 0x01
A relative value indicating the performance of the link between two devices
CU_DEVICE_P2P_ATTRIBUTE_ACCESS_SUPPORTED = 0x02
P2P Access is enable
CU_DEVICE_P2P_ATTRIBUTE_NATIVE_ATOMIC_SUPPORTED = 0x03
Atomic operation over the link supported
CU_DEVICE_P2P_ATTRIBUTE_ACCESS_ACCESS_SUPPORTED = 0x04
|
29
Modules
Deprecated use CU_DEVICE_P2P_ATTRIBUTE_CUDA_ARRAY_ACCESS_SUPPORTED
instead
CU_DEVICE_P2P_ATTRIBUTE_CUDA_ARRAY_ACCESS_SUPPORTED = 0x04
Accessing CUDA arrays over the link supported
enum CUdeviceNumaConfig
CUDA device NUMA configuration
Values
CU_DEVICE_NUMA_CONFIG_NONE = 0
The GPU is not a NUMA node
CU_DEVICE_NUMA_CONFIG_NUMA_NODE
The GPU is a NUMA node, CU_DEVICE_ATTRIBUTE_NUMA_ID contains its NUMA ID
enum CUdriverProcAddress_flags
Flags to specify search options. For more details see cuGetProcAddress
Values
CU_GET_PROC_ADDRESS_DEFAULT = 0
Default search mode for driver symbols.
CU_GET_PROC_ADDRESS_LEGACY_STREAM = 1<<0
Search for legacy versions of driver symbols.
CU_GET_PROC_ADDRESS_PER_THREAD_DEFAULT_STREAM = 1<<1
Search for per-thread versions of driver symbols.
enum CUdriverProcAddressQueryResult
Flags to indicate search status. For more details see cuGetProcAddress
Values
CU_GET_PROC_ADDRESS_SUCCESS = 0
Symbol was succesfully found
CU_GET_PROC_ADDRESS_SYMBOL_NOT_FOUND = 1
Symbol was not found in search
CU_GET_PROC_ADDRESS_VERSION_NOT_SUFFICIENT = 2
Symbol was found but version supplied was not sufficient
|
30
Modules
enum CUeglColorFormat
CUDA EGL Color Format - The different planar and multiplanar formats currently
supported for CUDA_EGL interops. Three channel formats are currently not supported for
CU_EGL_FRAME_TYPE_ARRAY
Values
CU_EGL_COLOR_FORMAT_YUV420_PLANAR = 0x00
Y, U, V in three surfaces, each in a separate surface, U/V width = 1/2 Y width, U/V height = 1/2 Y
height.
CU_EGL_COLOR_FORMAT_YUV420_SEMIPLANAR = 0x01
Y, UV in two surfaces (UV as one surface) with VU byte ordering, width, height ratio same as
YUV420Planar.
CU_EGL_COLOR_FORMAT_YUV422_PLANAR = 0x02
Y, U, V each in a separate surface, U/V width = 1/2 Y width, U/V height = Y height.
CU_EGL_COLOR_FORMAT_YUV422_SEMIPLANAR = 0x03
Y, UV in two surfaces with VU byte ordering, width, height ratio same as YUV422Planar.
CU_EGL_COLOR_FORMAT_RGB = 0x04
R/G/B three channels in one surface with BGR byte ordering. Only pitch linear format supported.
CU_EGL_COLOR_FORMAT_BGR = 0x05
R/G/B three channels in one surface with RGB byte ordering. Only pitch linear format supported.
CU_EGL_COLOR_FORMAT_ARGB = 0x06
R/G/B/A four channels in one surface with BGRA byte ordering.
CU_EGL_COLOR_FORMAT_RGBA = 0x07
R/G/B/A four channels in one surface with ABGR byte ordering.
CU_EGL_COLOR_FORMAT_L = 0x08
single luminance channel in one surface.
CU_EGL_COLOR_FORMAT_R = 0x09
single color channel in one surface.
CU_EGL_COLOR_FORMAT_YUV444_PLANAR = 0x0A
Y, U, V in three surfaces, each in a separate surface, U/V width = Y width, U/V height = Y height.
CU_EGL_COLOR_FORMAT_YUV444_SEMIPLANAR = 0x0B
Y, UV in two surfaces (UV as one surface) with VU byte ordering, width, height ratio same as
YUV444Planar.
CU_EGL_COLOR_FORMAT_YUYV_422 = 0x0C
Y, U, V in one surface, interleaved as UYVY in one channel.
CU_EGL_COLOR_FORMAT_UYVY_422 = 0x0D
Y, U, V in one surface, interleaved as YUYV in one channel.
CU_EGL_COLOR_FORMAT_ABGR = 0x0E
R/G/B/A four channels in one surface with RGBA byte ordering.
CU_EGL_COLOR_FORMAT_BGRA = 0x0F
R/G/B/A four channels in one surface with ARGB byte ordering.
|
31
Modules
CU_EGL_COLOR_FORMAT_A = 0x10
Alpha color format - one channel in one surface.
CU_EGL_COLOR_FORMAT_RG = 0x11
R/G color format - two channels in one surface with GR byte ordering
CU_EGL_COLOR_FORMAT_AYUV = 0x12
Y, U, V, A four channels in one surface, interleaved as VUYA.
CU_EGL_COLOR_FORMAT_YVU444_SEMIPLANAR = 0x13
Y, VU in two surfaces (VU as one surface) with UV byte ordering, U/V width = Y width, U/V
height = Y height.
CU_EGL_COLOR_FORMAT_YVU422_SEMIPLANAR = 0x14
Y, VU in two surfaces (VU as one surface) with UV byte ordering, U/V width = 1/2 Y width, U/V
height = Y height.
CU_EGL_COLOR_FORMAT_YVU420_SEMIPLANAR = 0x15
Y, VU in two surfaces (VU as one surface) with UV byte ordering, U/V width = 1/2 Y width, U/V
height = 1/2 Y height.
CU_EGL_COLOR_FORMAT_Y10V10U10_444_SEMIPLANAR = 0x16
Y10, V10U10 in two surfaces (VU as one surface) with UV byte ordering, U/V width = Y width, U/
V height = Y height.
CU_EGL_COLOR_FORMAT_Y10V10U10_420_SEMIPLANAR = 0x17
Y10, V10U10 in two surfaces (VU as one surface) with UV byte ordering, U/V width = 1/2 Y
width, U/V height = 1/2 Y height.
CU_EGL_COLOR_FORMAT_Y12V12U12_444_SEMIPLANAR = 0x18
Y12, V12U12 in two surfaces (VU as one surface) with UV byte ordering, U/V width = Y width, U/
V height = Y height.
CU_EGL_COLOR_FORMAT_Y12V12U12_420_SEMIPLANAR = 0x19
Y12, V12U12 in two surfaces (VU as one surface) with UV byte ordering, U/V width = 1/2 Y
width, U/V height = 1/2 Y height.
CU_EGL_COLOR_FORMAT_VYUY_ER = 0x1A
Extended Range Y, U, V in one surface, interleaved as YVYU in one channel.
CU_EGL_COLOR_FORMAT_UYVY_ER = 0x1B
Extended Range Y, U, V in one surface, interleaved as YUYV in one channel.
CU_EGL_COLOR_FORMAT_YUYV_ER = 0x1C
Extended Range Y, U, V in one surface, interleaved as UYVY in one channel.
CU_EGL_COLOR_FORMAT_YVYU_ER = 0x1D
Extended Range Y, U, V in one surface, interleaved as VYUY in one channel.
CU_EGL_COLOR_FORMAT_YUV_ER = 0x1E
Extended Range Y, U, V three channels in one surface, interleaved as VUY. Only pitch linear
format supported.
CU_EGL_COLOR_FORMAT_YUVA_ER = 0x1F
Extended Range Y, U, V, A four channels in one surface, interleaved as AVUY.
CU_EGL_COLOR_FORMAT_AYUV_ER = 0x20
Extended Range Y, U, V, A four channels in one surface, interleaved as VUYA.
CU_EGL_COLOR_FORMAT_YUV444_PLANAR_ER = 0x21
|
32
Modules
Extended Range Y, U, V in three surfaces, U/V width = Y width, U/V height = Y height.
CU_EGL_COLOR_FORMAT_YUV422_PLANAR_ER = 0x22
Extended Range Y, U, V in three surfaces, U/V width = 1/2 Y width, U/V height = Y height.
CU_EGL_COLOR_FORMAT_YUV420_PLANAR_ER = 0x23
Extended Range Y, U, V in three surfaces, U/V width = 1/2 Y width, U/V height = 1/2 Y height.
CU_EGL_COLOR_FORMAT_YUV444_SEMIPLANAR_ER = 0x24
Extended Range Y, UV in two surfaces (UV as one surface) with VU byte ordering, U/V width = Y
width, U/V height = Y height.
CU_EGL_COLOR_FORMAT_YUV422_SEMIPLANAR_ER = 0x25
Extended Range Y, UV in two surfaces (UV as one surface) with VU byte ordering, U/V width =
1/2 Y width, U/V height = Y height.
CU_EGL_COLOR_FORMAT_YUV420_SEMIPLANAR_ER = 0x26
Extended Range Y, UV in two surfaces (UV as one surface) with VU byte ordering, U/V width =
1/2 Y width, U/V height = 1/2 Y height.
CU_EGL_COLOR_FORMAT_YVU444_PLANAR_ER = 0x27
Extended Range Y, V, U in three surfaces, U/V width = Y width, U/V height = Y height.
CU_EGL_COLOR_FORMAT_YVU422_PLANAR_ER = 0x28
Extended Range Y, V, U in three surfaces, U/V width = 1/2 Y width, U/V height = Y height.
CU_EGL_COLOR_FORMAT_YVU420_PLANAR_ER = 0x29
Extended Range Y, V, U in three surfaces, U/V width = 1/2 Y width, U/V height = 1/2 Y height.
CU_EGL_COLOR_FORMAT_YVU444_SEMIPLANAR_ER = 0x2A
Extended Range Y, VU in two surfaces (VU as one surface) with UV byte ordering, U/V width = Y
width, U/V height = Y height.
CU_EGL_COLOR_FORMAT_YVU422_SEMIPLANAR_ER = 0x2B
Extended Range Y, VU in two surfaces (VU as one surface) with UV byte ordering, U/V width =
1/2 Y width, U/V height = Y height.
CU_EGL_COLOR_FORMAT_YVU420_SEMIPLANAR_ER = 0x2C
Extended Range Y, VU in two surfaces (VU as one surface) with UV byte ordering, U/V width =
1/2 Y width, U/V height = 1/2 Y height.
CU_EGL_COLOR_FORMAT_BAYER_RGGB = 0x2D
Bayer format - one channel in one surface with interleaved RGGB ordering.
CU_EGL_COLOR_FORMAT_BAYER_BGGR = 0x2E
Bayer format - one channel in one surface with interleaved BGGR ordering.
CU_EGL_COLOR_FORMAT_BAYER_GRBG = 0x2F
Bayer format - one channel in one surface with interleaved GRBG ordering.
CU_EGL_COLOR_FORMAT_BAYER_GBRG = 0x30
Bayer format - one channel in one surface with interleaved GBRG ordering.
CU_EGL_COLOR_FORMAT_BAYER10_RGGB = 0x31
Bayer10 format - one channel in one surface with interleaved RGGB ordering. Out of 16 bits, 10
bits used 6 bits No-op.
CU_EGL_COLOR_FORMAT_BAYER10_BGGR = 0x32
Bayer10 format - one channel in one surface with interleaved BGGR ordering. Out of 16 bits, 10
bits used 6 bits No-op.
|
33
Modules
CU_EGL_COLOR_FORMAT_BAYER10_GRBG = 0x33
Bayer10 format - one channel in one surface with interleaved GRBG ordering. Out of 16 bits, 10
bits used 6 bits No-op.
CU_EGL_COLOR_FORMAT_BAYER10_GBRG = 0x34
Bayer10 format - one channel in one surface with interleaved GBRG ordering. Out of 16 bits, 10
bits used 6 bits No-op.
CU_EGL_COLOR_FORMAT_BAYER12_RGGB = 0x35
Bayer12 format - one channel in one surface with interleaved RGGB ordering. Out of 16 bits, 12
bits used 4 bits No-op.
CU_EGL_COLOR_FORMAT_BAYER12_BGGR = 0x36
Bayer12 format - one channel in one surface with interleaved BGGR ordering. Out of 16 bits, 12
bits used 4 bits No-op.
CU_EGL_COLOR_FORMAT_BAYER12_GRBG = 0x37
Bayer12 format - one channel in one surface with interleaved GRBG ordering. Out of 16 bits, 12
bits used 4 bits No-op.
CU_EGL_COLOR_FORMAT_BAYER12_GBRG = 0x38
Bayer12 format - one channel in one surface with interleaved GBRG ordering. Out of 16 bits, 12
bits used 4 bits No-op.
CU_EGL_COLOR_FORMAT_BAYER14_RGGB = 0x39
Bayer14 format - one channel in one surface with interleaved RGGB ordering. Out of 16 bits, 14
bits used 2 bits No-op.
CU_EGL_COLOR_FORMAT_BAYER14_BGGR = 0x3A
Bayer14 format - one channel in one surface with interleaved BGGR ordering. Out of 16 bits, 14
bits used 2 bits No-op.
CU_EGL_COLOR_FORMAT_BAYER14_GRBG = 0x3B
Bayer14 format - one channel in one surface with interleaved GRBG ordering. Out of 16 bits, 14
bits used 2 bits No-op.
CU_EGL_COLOR_FORMAT_BAYER14_GBRG = 0x3C
Bayer14 format - one channel in one surface with interleaved GBRG ordering. Out of 16 bits, 14
bits used 2 bits No-op.
CU_EGL_COLOR_FORMAT_BAYER20_RGGB = 0x3D
Bayer20 format - one channel in one surface with interleaved RGGB ordering. Out of 32 bits, 20
bits used 12 bits No-op.
CU_EGL_COLOR_FORMAT_BAYER20_BGGR = 0x3E
Bayer20 format - one channel in one surface with interleaved BGGR ordering. Out of 32 bits, 20
bits used 12 bits No-op.
CU_EGL_COLOR_FORMAT_BAYER20_GRBG = 0x3F
Bayer20 format - one channel in one surface with interleaved GRBG ordering. Out of 32 bits, 20
bits used 12 bits No-op.
CU_EGL_COLOR_FORMAT_BAYER20_GBRG = 0x40
Bayer20 format - one channel in one surface with interleaved GBRG ordering. Out of 32 bits, 20
bits used 12 bits No-op.
CU_EGL_COLOR_FORMAT_YVU444_PLANAR = 0x41
|
34
Modules
Y, V, U in three surfaces, each in a separate surface, U/V width = Y width, U/V height = Y height.
CU_EGL_COLOR_FORMAT_YVU422_PLANAR = 0x42
Y, V, U in three surfaces, each in a separate surface, U/V width = 1/2 Y width, U/V height = Y
height.
CU_EGL_COLOR_FORMAT_YVU420_PLANAR = 0x43
Y, V, U in three surfaces, each in a separate surface, U/V width = 1/2 Y width, U/V height = 1/2 Y
height.
CU_EGL_COLOR_FORMAT_BAYER_ISP_RGGB = 0x44
Nvidia proprietary Bayer ISP format - one channel in one surface with interleaved RGGB ordering
and mapped to opaque integer datatype.
CU_EGL_COLOR_FORMAT_BAYER_ISP_BGGR = 0x45
Nvidia proprietary Bayer ISP format - one channel in one surface with interleaved BGGR ordering
and mapped to opaque integer datatype.
CU_EGL_COLOR_FORMAT_BAYER_ISP_GRBG = 0x46
Nvidia proprietary Bayer ISP format - one channel in one surface with interleaved GRBG ordering
and mapped to opaque integer datatype.
CU_EGL_COLOR_FORMAT_BAYER_ISP_GBRG = 0x47
Nvidia proprietary Bayer ISP format - one channel in one surface with interleaved GBRG ordering
and mapped to opaque integer datatype.
CU_EGL_COLOR_FORMAT_BAYER_BCCR = 0x48
Bayer format - one channel in one surface with interleaved BCCR ordering.
CU_EGL_COLOR_FORMAT_BAYER_RCCB = 0x49
Bayer format - one channel in one surface with interleaved RCCB ordering.
CU_EGL_COLOR_FORMAT_BAYER_CRBC = 0x4A
Bayer format - one channel in one surface with interleaved CRBC ordering.
CU_EGL_COLOR_FORMAT_BAYER_CBRC = 0x4B
Bayer format - one channel in one surface with interleaved CBRC ordering.
CU_EGL_COLOR_FORMAT_BAYER10_CCCC = 0x4C
Bayer10 format - one channel in one surface with interleaved CCCC ordering. Out of 16 bits, 10
bits used 6 bits No-op.
CU_EGL_COLOR_FORMAT_BAYER12_BCCR = 0x4D
Bayer12 format - one channel in one surface with interleaved BCCR ordering. Out of 16 bits, 12
bits used 4 bits No-op.
CU_EGL_COLOR_FORMAT_BAYER12_RCCB = 0x4E
Bayer12 format - one channel in one surface with interleaved RCCB ordering. Out of 16 bits, 12
bits used 4 bits No-op.
CU_EGL_COLOR_FORMAT_BAYER12_CRBC = 0x4F
Bayer12 format - one channel in one surface with interleaved CRBC ordering. Out of 16 bits, 12
bits used 4 bits No-op.
CU_EGL_COLOR_FORMAT_BAYER12_CBRC = 0x50
Bayer12 format - one channel in one surface with interleaved CBRC ordering. Out of 16 bits, 12
bits used 4 bits No-op.
CU_EGL_COLOR_FORMAT_BAYER12_CCCC = 0x51
|
35
Modules
Bayer12 format - one channel in one surface with interleaved CCCC ordering. Out of 16 bits, 12
bits used 4 bits No-op.
CU_EGL_COLOR_FORMAT_Y = 0x52
Color format for single Y plane.
CU_EGL_COLOR_FORMAT_YUV420_SEMIPLANAR_2020 = 0x53
Y, UV in two surfaces (UV as one surface) U/V width = 1/2 Y width, U/V height = 1/2 Y height.
CU_EGL_COLOR_FORMAT_YVU420_SEMIPLANAR_2020 = 0x54
Y, VU in two surfaces (VU as one surface) U/V width = 1/2 Y width, U/V height = 1/2 Y height.
CU_EGL_COLOR_FORMAT_YUV420_PLANAR_2020 = 0x55
Y, U, V each in a separate surface, U/V width = 1/2 Y width, U/V height= 1/2 Y height.
CU_EGL_COLOR_FORMAT_YVU420_PLANAR_2020 = 0x56
Y, V, U each in a separate surface, U/V width = 1/2 Y width, U/V height = 1/2 Y height.
CU_EGL_COLOR_FORMAT_YUV420_SEMIPLANAR_709 = 0x57
Y, UV in two surfaces (UV as one surface) U/V width = 1/2 Y width, U/V height = 1/2 Y height.
CU_EGL_COLOR_FORMAT_YVU420_SEMIPLANAR_709 = 0x58
Y, VU in two surfaces (VU as one surface) U/V width = 1/2 Y width, U/V height = 1/2 Y height.
CU_EGL_COLOR_FORMAT_YUV420_PLANAR_709 = 0x59
Y, U, V each in a separate surface, U/V width = 1/2 Y width, U/V height = 1/2 Y height.
CU_EGL_COLOR_FORMAT_YVU420_PLANAR_709 = 0x5A
Y, V, U each in a separate surface, U/V width = 1/2 Y width, U/V height = 1/2 Y height.
CU_EGL_COLOR_FORMAT_Y10V10U10_420_SEMIPLANAR_709 = 0x5B
Y10, V10U10 in two surfaces (VU as one surface), U/V width = 1/2 Y width, U/V height = 1/2 Y
height.
CU_EGL_COLOR_FORMAT_Y10V10U10_420_SEMIPLANAR_2020 = 0x5C
Y10, V10U10 in two surfaces (VU as one surface), U/V width = 1/2 Y width, U/V height = 1/2 Y
height.
CU_EGL_COLOR_FORMAT_Y10V10U10_422_SEMIPLANAR_2020 = 0x5D
Y10, V10U10 in two surfaces(VU as one surface) U/V width = 1/2 Y width, U/V height = Y height.
CU_EGL_COLOR_FORMAT_Y10V10U10_422_SEMIPLANAR = 0x5E
Y10, V10U10 in two surfaces(VU as one surface) U/V width = 1/2 Y width, U/V height = Y height.
CU_EGL_COLOR_FORMAT_Y10V10U10_422_SEMIPLANAR_709 = 0x5F
Y10, V10U10 in two surfaces(VU as one surface) U/V width = 1/2 Y width, U/V height = Y height.
CU_EGL_COLOR_FORMAT_Y_ER = 0x60
Extended Range Color format for single Y plane.
CU_EGL_COLOR_FORMAT_Y_709_ER = 0x61
Extended Range Color format for single Y plane.
CU_EGL_COLOR_FORMAT_Y10_ER = 0x62
Extended Range Color format for single Y10 plane.
CU_EGL_COLOR_FORMAT_Y10_709_ER = 0x63
Extended Range Color format for single Y10 plane.
CU_EGL_COLOR_FORMAT_Y12_ER = 0x64
Extended Range Color format for single Y12 plane.
CU_EGL_COLOR_FORMAT_Y12_709_ER = 0x65
|
36
////////////////////////////////////////// |
||
|
|
|