cuda 0.11.0.1 → 0.12.8.0
raw patch · 13 files changed
+282/−609 lines, 13 filesdep +containerssetup-changednew-uploaderPVP ok
version bump matches the API change (PVP)
Dependencies added: containers
API changes (from Hackage documentation)
- Foreign.CUDA.Driver.Device: CanUse64BitStreamMemOpsV2 :: DeviceAttribute
- Foreign.CUDA.Driver.Device: CanUseStreamMemOps :: DeviceAttribute
- Foreign.CUDA.Driver.Device: CanUseStreamWaitValueNorV2 :: DeviceAttribute
- Foreign.CUDA.Driver.Module: Compute20 :: JITTarget
- Foreign.CUDA.Driver.Module: Compute21 :: JITTarget
- Foreign.CUDA.Driver.Module.Base: Compute20 :: JITTarget
- Foreign.CUDA.Driver.Module.Base: Compute21 :: JITTarget
- Foreign.CUDA.Driver.Module.Base: loadFile :: FilePath -> IO Module
- Foreign.CUDA.Driver.Module.Query: getTex :: Module -> ShortByteString -> IO Texture
- Foreign.CUDA.Driver.Texture: Bc1Unorm :: Format
- Foreign.CUDA.Driver.Texture: Bc1UnormSrgb :: Format
- Foreign.CUDA.Driver.Texture: Bc2Unorm :: Format
- Foreign.CUDA.Driver.Texture: Bc2UnormSrgb :: Format
- Foreign.CUDA.Driver.Texture: Bc3Unorm :: Format
- Foreign.CUDA.Driver.Texture: Bc3UnormSrgb :: Format
- Foreign.CUDA.Driver.Texture: Bc4Snorm :: Format
- Foreign.CUDA.Driver.Texture: Bc4Unorm :: Format
- Foreign.CUDA.Driver.Texture: Bc5Snorm :: Format
- Foreign.CUDA.Driver.Texture: Bc5Unorm :: Format
- Foreign.CUDA.Driver.Texture: Bc6hSf16 :: Format
- Foreign.CUDA.Driver.Texture: Bc6hUf16 :: Format
- Foreign.CUDA.Driver.Texture: Bc7Unorm :: Format
- Foreign.CUDA.Driver.Texture: Bc7UnormSrgb :: Format
- Foreign.CUDA.Driver.Texture: Border :: AddressMode
- Foreign.CUDA.Driver.Texture: Clamp :: AddressMode
- Foreign.CUDA.Driver.Texture: Float :: Format
- Foreign.CUDA.Driver.Texture: Half :: Format
- Foreign.CUDA.Driver.Texture: Int16 :: Format
- Foreign.CUDA.Driver.Texture: Int32 :: Format
- Foreign.CUDA.Driver.Texture: Int8 :: Format
- Foreign.CUDA.Driver.Texture: Linear :: FilterMode
- Foreign.CUDA.Driver.Texture: Mirror :: AddressMode
- Foreign.CUDA.Driver.Texture: NormalizedCoordinates :: ReadMode
- Foreign.CUDA.Driver.Texture: Nv12 :: Format
- Foreign.CUDA.Driver.Texture: Point :: FilterMode
- Foreign.CUDA.Driver.Texture: ReadAsInteger :: ReadMode
- Foreign.CUDA.Driver.Texture: SRGB :: ReadMode
- Foreign.CUDA.Driver.Texture: SnormInt16x1 :: Format
- Foreign.CUDA.Driver.Texture: SnormInt16x2 :: Format
- Foreign.CUDA.Driver.Texture: SnormInt16x4 :: Format
- Foreign.CUDA.Driver.Texture: SnormInt8x1 :: Format
- Foreign.CUDA.Driver.Texture: SnormInt8x2 :: Format
- Foreign.CUDA.Driver.Texture: SnormInt8x4 :: Format
- Foreign.CUDA.Driver.Texture: Texture :: Ptr () -> Texture
- Foreign.CUDA.Driver.Texture: UnormInt16x1 :: Format
- Foreign.CUDA.Driver.Texture: UnormInt16x2 :: Format
- Foreign.CUDA.Driver.Texture: UnormInt16x4 :: Format
- Foreign.CUDA.Driver.Texture: UnormInt8x1 :: Format
- Foreign.CUDA.Driver.Texture: UnormInt8x2 :: Format
- Foreign.CUDA.Driver.Texture: UnormInt8x4 :: Format
- Foreign.CUDA.Driver.Texture: Word16 :: Format
- Foreign.CUDA.Driver.Texture: Word32 :: Format
- Foreign.CUDA.Driver.Texture: Word8 :: Format
- Foreign.CUDA.Driver.Texture: Wrap :: AddressMode
- Foreign.CUDA.Driver.Texture: [useTexture] :: Texture -> Ptr ()
- Foreign.CUDA.Driver.Texture: bind :: Texture -> DevicePtr a -> Int64 -> IO ()
- Foreign.CUDA.Driver.Texture: bind2D :: Texture -> Format -> Int -> DevicePtr a -> (Int, Int) -> Int64 -> IO ()
- Foreign.CUDA.Driver.Texture: create :: IO Texture
- Foreign.CUDA.Driver.Texture: data AddressMode
- Foreign.CUDA.Driver.Texture: data FilterMode
- Foreign.CUDA.Driver.Texture: data Format
- Foreign.CUDA.Driver.Texture: data ReadMode
- Foreign.CUDA.Driver.Texture: getAddressMode :: Texture -> Int -> IO AddressMode
- Foreign.CUDA.Driver.Texture: getFilterMode :: Texture -> IO FilterMode
- Foreign.CUDA.Driver.Texture: getFormat :: Texture -> IO (Format, Int)
- Foreign.CUDA.Driver.Texture: instance Foreign.Storable.Storable Foreign.CUDA.Driver.Texture.Texture
- Foreign.CUDA.Driver.Texture: instance GHC.Classes.Eq Foreign.CUDA.Driver.Texture.AddressMode
- Foreign.CUDA.Driver.Texture: instance GHC.Classes.Eq Foreign.CUDA.Driver.Texture.FilterMode
- Foreign.CUDA.Driver.Texture: instance GHC.Classes.Eq Foreign.CUDA.Driver.Texture.Format
- Foreign.CUDA.Driver.Texture: instance GHC.Classes.Eq Foreign.CUDA.Driver.Texture.ReadMode
- Foreign.CUDA.Driver.Texture: instance GHC.Classes.Eq Foreign.CUDA.Driver.Texture.Texture
- Foreign.CUDA.Driver.Texture: instance GHC.Enum.Enum Foreign.CUDA.Driver.Texture.AddressMode
- Foreign.CUDA.Driver.Texture: instance GHC.Enum.Enum Foreign.CUDA.Driver.Texture.FilterMode
- Foreign.CUDA.Driver.Texture: instance GHC.Enum.Enum Foreign.CUDA.Driver.Texture.Format
- Foreign.CUDA.Driver.Texture: instance GHC.Enum.Enum Foreign.CUDA.Driver.Texture.ReadMode
- Foreign.CUDA.Driver.Texture: instance GHC.Show.Show Foreign.CUDA.Driver.Texture.AddressMode
- Foreign.CUDA.Driver.Texture: instance GHC.Show.Show Foreign.CUDA.Driver.Texture.FilterMode
- Foreign.CUDA.Driver.Texture: instance GHC.Show.Show Foreign.CUDA.Driver.Texture.Format
- Foreign.CUDA.Driver.Texture: instance GHC.Show.Show Foreign.CUDA.Driver.Texture.ReadMode
- Foreign.CUDA.Driver.Texture: instance GHC.Show.Show Foreign.CUDA.Driver.Texture.Texture
- Foreign.CUDA.Driver.Texture: newtype Texture
- Foreign.CUDA.Driver.Texture: setAddressMode :: Texture -> Int -> AddressMode -> IO ()
- Foreign.CUDA.Driver.Texture: setFilterMode :: Texture -> FilterMode -> IO ()
- Foreign.CUDA.Driver.Texture: setFormat :: Texture -> Format -> Int -> IO ()
- Foreign.CUDA.Driver.Texture: setReadMode :: Texture -> ReadMode -> IO ()
- Foreign.CUDA.Runtime.Texture: Border :: AddressMode
- Foreign.CUDA.Runtime.Texture: Clamp :: AddressMode
- Foreign.CUDA.Runtime.Texture: Float :: FormatKind
- Foreign.CUDA.Runtime.Texture: FormatDesc :: !(Int, Int, Int, Int) -> !FormatKind -> FormatDesc
- Foreign.CUDA.Runtime.Texture: Linear :: FilterMode
- Foreign.CUDA.Runtime.Texture: Mirror :: AddressMode
- Foreign.CUDA.Runtime.Texture: NV12 :: FormatKind
- Foreign.CUDA.Runtime.Texture: None :: FormatKind
- Foreign.CUDA.Runtime.Texture: Point :: FilterMode
- Foreign.CUDA.Runtime.Texture: Signed :: FormatKind
- Foreign.CUDA.Runtime.Texture: SignedBlockCompressed4 :: FormatKind
- Foreign.CUDA.Runtime.Texture: SignedBlockCompressed5 :: FormatKind
- Foreign.CUDA.Runtime.Texture: SignedBlockCompressed6H :: FormatKind
- Foreign.CUDA.Runtime.Texture: SignedNormalized16X1 :: FormatKind
- Foreign.CUDA.Runtime.Texture: SignedNormalized16X2 :: FormatKind
- Foreign.CUDA.Runtime.Texture: SignedNormalized16X4 :: FormatKind
- Foreign.CUDA.Runtime.Texture: SignedNormalized8X1 :: FormatKind
- Foreign.CUDA.Runtime.Texture: SignedNormalized8X2 :: FormatKind
- Foreign.CUDA.Runtime.Texture: SignedNormalized8X4 :: FormatKind
- Foreign.CUDA.Runtime.Texture: Texture :: !Bool -> !FilterMode -> !(AddressMode, AddressMode, AddressMode) -> !FormatDesc -> Texture
- Foreign.CUDA.Runtime.Texture: Unsigned :: FormatKind
- Foreign.CUDA.Runtime.Texture: UnsignedBlockCompressed1 :: FormatKind
- Foreign.CUDA.Runtime.Texture: UnsignedBlockCompressed1SRGB :: FormatKind
- Foreign.CUDA.Runtime.Texture: UnsignedBlockCompressed2 :: FormatKind
- Foreign.CUDA.Runtime.Texture: UnsignedBlockCompressed2SRGB :: FormatKind
- Foreign.CUDA.Runtime.Texture: UnsignedBlockCompressed3 :: FormatKind
- Foreign.CUDA.Runtime.Texture: UnsignedBlockCompressed3SRGB :: FormatKind
- Foreign.CUDA.Runtime.Texture: UnsignedBlockCompressed4 :: FormatKind
- Foreign.CUDA.Runtime.Texture: UnsignedBlockCompressed5 :: FormatKind
- Foreign.CUDA.Runtime.Texture: UnsignedBlockCompressed6H :: FormatKind
- Foreign.CUDA.Runtime.Texture: UnsignedBlockCompressed7 :: FormatKind
- Foreign.CUDA.Runtime.Texture: UnsignedBlockCompressed7SRGB :: FormatKind
- Foreign.CUDA.Runtime.Texture: UnsignedNormalized16X1 :: FormatKind
- Foreign.CUDA.Runtime.Texture: UnsignedNormalized16X2 :: FormatKind
- Foreign.CUDA.Runtime.Texture: UnsignedNormalized16X4 :: FormatKind
- Foreign.CUDA.Runtime.Texture: UnsignedNormalized8X1 :: FormatKind
- Foreign.CUDA.Runtime.Texture: UnsignedNormalized8X2 :: FormatKind
- Foreign.CUDA.Runtime.Texture: UnsignedNormalized8X4 :: FormatKind
- Foreign.CUDA.Runtime.Texture: Wrap :: AddressMode
- Foreign.CUDA.Runtime.Texture: [addressing] :: Texture -> !(AddressMode, AddressMode, AddressMode)
- Foreign.CUDA.Runtime.Texture: [depth] :: FormatDesc -> !(Int, Int, Int, Int)
- Foreign.CUDA.Runtime.Texture: [filtering] :: Texture -> !FilterMode
- Foreign.CUDA.Runtime.Texture: [format] :: Texture -> !FormatDesc
- Foreign.CUDA.Runtime.Texture: [kind] :: FormatDesc -> !FormatKind
- Foreign.CUDA.Runtime.Texture: [normalised] :: Texture -> !Bool
- Foreign.CUDA.Runtime.Texture: bind :: String -> Texture -> DevicePtr a -> Int64 -> IO ()
- Foreign.CUDA.Runtime.Texture: bind2D :: String -> Texture -> DevicePtr a -> (Int, Int) -> Int64 -> IO ()
- Foreign.CUDA.Runtime.Texture: data AddressMode
- Foreign.CUDA.Runtime.Texture: data FilterMode
- Foreign.CUDA.Runtime.Texture: data FormatDesc
- Foreign.CUDA.Runtime.Texture: data FormatKind
- Foreign.CUDA.Runtime.Texture: data Texture
- Foreign.CUDA.Runtime.Texture: instance Foreign.Storable.Storable Foreign.CUDA.Runtime.Texture.FormatDesc
- Foreign.CUDA.Runtime.Texture: instance Foreign.Storable.Storable Foreign.CUDA.Runtime.Texture.Texture
- Foreign.CUDA.Runtime.Texture: instance GHC.Classes.Eq Foreign.CUDA.Runtime.Texture.AddressMode
- Foreign.CUDA.Runtime.Texture: instance GHC.Classes.Eq Foreign.CUDA.Runtime.Texture.FilterMode
- Foreign.CUDA.Runtime.Texture: instance GHC.Classes.Eq Foreign.CUDA.Runtime.Texture.FormatDesc
- Foreign.CUDA.Runtime.Texture: instance GHC.Classes.Eq Foreign.CUDA.Runtime.Texture.FormatKind
- Foreign.CUDA.Runtime.Texture: instance GHC.Classes.Eq Foreign.CUDA.Runtime.Texture.Texture
- Foreign.CUDA.Runtime.Texture: instance GHC.Enum.Enum Foreign.CUDA.Runtime.Texture.AddressMode
- Foreign.CUDA.Runtime.Texture: instance GHC.Enum.Enum Foreign.CUDA.Runtime.Texture.FilterMode
- Foreign.CUDA.Runtime.Texture: instance GHC.Enum.Enum Foreign.CUDA.Runtime.Texture.FormatKind
- Foreign.CUDA.Runtime.Texture: instance GHC.Show.Show Foreign.CUDA.Runtime.Texture.AddressMode
- Foreign.CUDA.Runtime.Texture: instance GHC.Show.Show Foreign.CUDA.Runtime.Texture.FilterMode
- Foreign.CUDA.Runtime.Texture: instance GHC.Show.Show Foreign.CUDA.Runtime.Texture.FormatDesc
- Foreign.CUDA.Runtime.Texture: instance GHC.Show.Show Foreign.CUDA.Runtime.Texture.FormatKind
- Foreign.CUDA.Runtime.Texture: instance GHC.Show.Show Foreign.CUDA.Runtime.Texture.Texture
+ Foreign.CUDA.Driver: CigEnabled :: Limit
+ Foreign.CUDA.Driver: CigShmemFallbackEnabled :: Limit
+ Foreign.CUDA.Driver: CoredumpEnable :: ContextFlag
+ Foreign.CUDA.Driver: OnlyPartialNativeAtomicSupported :: PeerAttribute
+ Foreign.CUDA.Driver: ShmemSize :: Limit
+ Foreign.CUDA.Driver: SyncMemops :: ContextFlag
+ Foreign.CUDA.Driver: UserCoredumpEnable :: ContextFlag
+ Foreign.CUDA.Driver.Context.Base: CoredumpEnable :: ContextFlag
+ Foreign.CUDA.Driver.Context.Base: SyncMemops :: ContextFlag
+ Foreign.CUDA.Driver.Context.Base: UserCoredumpEnable :: ContextFlag
+ Foreign.CUDA.Driver.Context.Config: CigEnabled :: Limit
+ Foreign.CUDA.Driver.Context.Config: CigShmemFallbackEnabled :: Limit
+ Foreign.CUDA.Driver.Context.Config: ShmemSize :: Limit
+ Foreign.CUDA.Driver.Context.Peer: OnlyPartialNativeAtomicSupported :: PeerAttribute
+ Foreign.CUDA.Driver.Device: CanUse64BitStreamMemOpsV1 :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: CanUseStreamMemOpsV1 :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: CanUseStreamWaitValueNorV1 :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: D3d12CigSupported :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: GpuPciDeviceId :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: GpuPciSubsystemId :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: HandleTypeFabricSupported :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: HostNumaId :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: HostNumaMemoryPoolsSupported :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: HostNumaMultinodeIpcSupported :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: HostNumaVirtualMemoryManagementSupported :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: IpcEventSupported :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: MemDecompressAlgorithmMask :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: MemDecompressMaximumLength :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: MemSyncDomainCount :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: MpsEnabled :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: MulticastSupported :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: NumaConfig :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: NumaId :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: TensorMapAccessSupported :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: UnifiedFunctionPointers :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: VulkanCigSupported :: DeviceAttribute
+ Foreign.CUDA.Driver.Error: CdpNotSupported :: Status
+ Foreign.CUDA.Driver.Error: CdpVersionMismatch :: Status
+ Foreign.CUDA.Driver.Error: Contained :: Status
+ Foreign.CUDA.Driver.Error: FunctionNotLoaded :: Status
+ Foreign.CUDA.Driver.Error: InvalidResourceConfiguration :: Status
+ Foreign.CUDA.Driver.Error: InvalidResourceType :: Status
+ Foreign.CUDA.Driver.Error: KeyRotation :: Status
+ Foreign.CUDA.Driver.Error: LossyQuery :: Status
+ Foreign.CUDA.Driver.Error: TensorMemoryLeak :: Status
+ Foreign.CUDA.Driver.Error: UnsupportedDevsideSync :: Status
+ Foreign.CUDA.Driver.Graph.Base: Conditional :: NodeType
+ Foreign.CUDA.Driver.Graph.Build: Conditional :: NodeType
+ Foreign.CUDA.Driver.Module: Compute100 :: JITTarget
+ Foreign.CUDA.Driver.Module: Compute100a :: JITTarget
+ Foreign.CUDA.Driver.Module: Compute100f :: JITTarget
+ Foreign.CUDA.Driver.Module: Compute101 :: JITTarget
+ Foreign.CUDA.Driver.Module: Compute101a :: JITTarget
+ Foreign.CUDA.Driver.Module: Compute101f :: JITTarget
+ Foreign.CUDA.Driver.Module: Compute103 :: JITTarget
+ Foreign.CUDA.Driver.Module: Compute103a :: JITTarget
+ Foreign.CUDA.Driver.Module: Compute103f :: JITTarget
+ Foreign.CUDA.Driver.Module: Compute120 :: JITTarget
+ Foreign.CUDA.Driver.Module: Compute120a :: JITTarget
+ Foreign.CUDA.Driver.Module: Compute120f :: JITTarget
+ Foreign.CUDA.Driver.Module: Compute121 :: JITTarget
+ Foreign.CUDA.Driver.Module: Compute121a :: JITTarget
+ Foreign.CUDA.Driver.Module: Compute121f :: JITTarget
+ Foreign.CUDA.Driver.Module: Compute90a :: JITTarget
+ Foreign.CUDA.Driver.Module.Base: Compute100 :: JITTarget
+ Foreign.CUDA.Driver.Module.Base: Compute100a :: JITTarget
+ Foreign.CUDA.Driver.Module.Base: Compute100f :: JITTarget
+ Foreign.CUDA.Driver.Module.Base: Compute101 :: JITTarget
+ Foreign.CUDA.Driver.Module.Base: Compute101a :: JITTarget
+ Foreign.CUDA.Driver.Module.Base: Compute101f :: JITTarget
+ Foreign.CUDA.Driver.Module.Base: Compute103 :: JITTarget
+ Foreign.CUDA.Driver.Module.Base: Compute103a :: JITTarget
+ Foreign.CUDA.Driver.Module.Base: Compute103f :: JITTarget
+ Foreign.CUDA.Driver.Module.Base: Compute120 :: JITTarget
+ Foreign.CUDA.Driver.Module.Base: Compute120a :: JITTarget
+ Foreign.CUDA.Driver.Module.Base: Compute120f :: JITTarget
+ Foreign.CUDA.Driver.Module.Base: Compute121 :: JITTarget
+ Foreign.CUDA.Driver.Module.Base: Compute121a :: JITTarget
+ Foreign.CUDA.Driver.Module.Base: Compute121f :: JITTarget
+ Foreign.CUDA.Driver.Module.Base: Compute90a :: JITTarget
+ Foreign.CUDA.Driver.Module.Base: loadData :: ByteString -> IO Module
+ Foreign.CUDA.Runtime.Error: CdpNotSupported :: Status
+ Foreign.CUDA.Runtime.Error: CdpVersionMismatch :: Status
+ Foreign.CUDA.Runtime.Error: Contained :: Status
+ Foreign.CUDA.Runtime.Error: FunctionNotLoaded :: Status
+ Foreign.CUDA.Runtime.Error: InvalidResourceConfiguration :: Status
+ Foreign.CUDA.Runtime.Error: InvalidResourceType :: Status
+ Foreign.CUDA.Runtime.Error: LossyQuery :: Status
+ Foreign.CUDA.Runtime.Error: TensorMemoryLeak :: Status
+ Foreign.CUDA.Runtime.Error: UnsupportedDevSideSync :: Status
Files
- CHANGELOG.md +13/−0
- README.md +26/−1
- Setup.hs +40/−25
- cbits/stubs.c +0/−16
- cuda.cabal +8/−22
- src/Foreign/CUDA/Analysis/Device.chs +131/−11
- src/Foreign/CUDA/Analysis/Occupancy.hs +8/−0
- src/Foreign/CUDA/Driver/Graph/Capture.chs +11/−1
- src/Foreign/CUDA/Driver/Module/Query.chs +1/−22
- src/Foreign/CUDA/Driver/Stream.chs +38/−0
- src/Foreign/CUDA/Driver/Texture.chs +0/−308
- src/Foreign/CUDA/Runtime/Device.chs +6/−0
- src/Foreign/CUDA/Runtime/Texture.chs +0/−203
CHANGELOG.md view
@@ -10,6 +10,19 @@ because NVIDIA are A-OK introducing breaking changes in minor updates. +## [0.12.8.0] - ???+### Added+ * Support for CUDA-12+ - Thanks to @noahmartinwilliams on GitHub for helping out!++### Removed+ * The following modules have been deprecated for a long time, and have+ finally been removed in CUDA-12:+ - `Foreign.CUDA.Driver.Texture`+ - `Foreign.CUDA.Runtime.Texture`+ Support for Texture Objects (their replacement) is missing in these+ bindings so far. Contributions welcome.+ ## [0.11.0.1] - 2023-08-15 ### Fixed * Build fixes for GHC 9.2 .. 9.6
README.md view
@@ -25,8 +25,13 @@ ## Missing functionality -An incomplete list of missing bindings. Pull requests welcome!+_This library is currently in **maintenance mode**. While we plan to release+updates to keep the existing interface working with newer CUDA versions (as+long as the underlying APIs remain available), no binding of new features is+planned at the moment. Get in touch if you want to contribute._ +Here is an incomplete historical list of missing bindings. Pull requests welcome!+ ### CUDA-9 - cuLaunchCooperativeKernelMultiDevice@@ -145,3 +150,23 @@ - cuGraphMemAllocNodeGetParams - cuGraphMemFreeNodeGetParams +### CUDA-12++A lot. PRs welcome.+++# Old compatibility notes++The setup script for this package requires at least Cabal-1.24. If you run into trouble with this:++* Cabal users: ensure you are using a new `cabal` executable and have run `cabal update` anywhere in the last few years. If you have previously run `cabal install` on libraries and have a broken environment as a result, remove `~/.ghc/<platfom>/environments/default`.+* Stack users: one may attempt @stack setup --upgrade-cabal@.++Due to an interaction between GHC-8 and unified virtual address spaces in+CUDA, this package does not currently work with GHCi on ghc-8.0.1 (compiled+programs should work). See the following for more details:++* <https://github.com/tmcdonell/cuda/issues/39>+* <https://ghc.haskell.org/trac/ghc/ticket/12573>++The bug should be fixed in ghc-8.0.2 and beyond.
Setup.hs view
@@ -36,6 +36,7 @@ import Control.Exception import Control.Monad+import Data.Char (isDigit) import Data.Function import Data.List import Data.Maybe@@ -140,7 +141,7 @@ -> IO HookedBuildInfo libraryBuildInfo verbosity profile installPath platform@(Platform arch os) ghcVersion extraLibs extraIncludes = do let- libraryPaths = cudaLibraryPath platform installPath : extraLibs+ libraryPaths = cudaLibraryPaths platform installPath ++ extraLibs includePaths = cudaIncludePath platform installPath : extraIncludes takeFirstExisting paths = do@@ -215,18 +216,19 @@ cudaIncludePath _ installPath = installPath </> "include" --- Return the location of the libraries relative to the base CUDA installation.+-- Return the potential locations of the libraries relative to the base CUDA installation. ---cudaLibraryPath :: Platform -> FilePath -> FilePath-cudaLibraryPath (Platform arch os) installPath = installPath </> libpath+cudaLibraryPaths :: Platform -> FilePath -> [FilePath]+cudaLibraryPaths (Platform arch os) installPath = [ installPath </> path | path <- libpaths ] where- libpath =+ libpaths = case (os, arch) of- (Windows, I386) -> "lib/Win32"- (Windows, X86_64) -> "lib/x64"- (OSX, _) -> "lib" -- MacOS does not distinguish 32- vs. 64-bit paths- (_, X86_64) -> "lib64" -- treat all others similarly- _ -> "lib"+ (Windows, I386) -> ["lib/Win32"]+ (Windows, X86_64) -> ["lib/x64"]+ (OSX, _) -> ["lib"] -- MacOS does not distinguish 32- vs. 64-bit paths+ (_, X86_64) -> ["lib64", "lib"] -- prefer lib64 for 64-bit systems+ (_, AArch64) -> ["lib64", "lib"]+ _ -> ["lib"] -- otherwise -- On Windows and OSX we use different libraries depending on whether we are@@ -264,7 +266,9 @@ -> [FilePath] -> IO [FilePath] cudaGhciLibrariesWindows platform installPath libraries = do- candidates <- mapM (importLibraryToDLLFileName platform) [ cudaLibraryPath platform installPath </> lib <.> "lib" | lib <- libraries ]+ candidates <- mapM (importLibraryToDLLFileName platform)+ [ libPath </> lib <.> "lib" | libPath <- cudaLibraryPaths platform installPath+ , lib <- libraries ] return [ dropExtension dll | Just dll <- candidates ] @@ -460,7 +464,7 @@ -> Platform -> IO FilePath findCUDAInstallPath verbosity platform = do- result <- findFirstValidLocation verbosity platform (candidateCUDAInstallPaths verbosity platform)+ result <- findFirstValidLocation verbosity platform =<< candidateCUDAInstallPaths verbosity platform case result of Just installPath -> do notice verbosity $ printf "Found CUDA toolkit at: %s (set CUDA_PATH to override this)" installPath@@ -547,19 +551,15 @@ candidateCUDAInstallPaths :: Verbosity -> Platform- -> [(IO FilePath, String)]-candidateCUDAInstallPaths verbosity platform =- [ (getEnv "CUDA_PATH", "environment variable CUDA_PATH")- , (findInPath, "nvcc compiler executable in PATH")- , (return defaultPath, printf "default install location (%s)" defaultPath)- , (getEnv "CUDA_PATH_V9_1", "environment variable CUDA_PATH_V9_1")- , (getEnv "CUDA_PATH_V9_0", "environment variable CUDA_PATH_V9_0")- , (getEnv "CUDA_PATH_V8_0", "environment variable CUDA_PATH_V8_0")- , (getEnv "CUDA_PATH_V7_5", "environment variable CUDA_PATH_V7_5")- , (getEnv "CUDA_PATH_V7_0", "environment variable CUDA_PATH_V7_0")- , (getEnv "CUDA_PATH_V6_5", "environment variable CUDA_PATH_V6_5")- , (getEnv "CUDA_PATH_V6_0", "environment variable CUDA_PATH_V6_0")- ]+ -> IO [(IO FilePath, String)]+candidateCUDAInstallPaths verbosity platform = do+ let defaults =+ [ (getEnv "CUDA_PATH", "environment variable CUDA_PATH")+ , (findInPath, "nvcc compiler executable in PATH")+ , (return defaultPath, printf "default install location (%s)" defaultPath)+ ]+ verVars <- versionedVars+ return $ defaults ++ verVars where findInPath :: IO FilePath findInPath = do@@ -570,6 +570,21 @@ defaultPath :: FilePath defaultPath = defaultCUDAInstallPath platform++ versionedVars :: IO [(IO FilePath, String)]+ versionedVars = do+ pairs <- getEnvironment+ let sorted = sort (mapMaybe (\(k, v) -> (,k,v) <$> parseCudaPathVerVar k) pairs)+ return [(return v, "environment variable " ++ k) | (_, k, v) <- sorted]++ parseCudaPathVerVar :: String -> Maybe (Int, Int)+ parseCudaPathVerVar var+ | ("CUDA_PATH_V", s1) <- splitAt 11 var+ , (n1, '_':n2) <- span isDigit s1, not (null n1)+ , all isDigit n2, not (null n2)+ = Just (read n1, read n2)+ | otherwise+ = Nothing -- NOTE: this function throws an exception when there is no `nvcc` in PATH.
cbits/stubs.c view
@@ -22,17 +22,6 @@ } #endif -CUresult cuTexRefSetAddress2D_simple(CUtexref tex, CUarray_format format, unsigned int numChannels, CUdeviceptr dptr, size_t width, size_t height, size_t pitch)-{- CUDA_ARRAY_DESCRIPTOR desc;- desc.Format = format;- desc.NumChannels = numChannels;- desc.Width = width;- desc.Height = height;-- return cuTexRefSetAddress2D(tex, &desc, dptr, pitch);-}- CUresult cuMemcpy2DHtoD(CUdeviceptr dstDevice, unsigned int dstPitch, unsigned int dstXInBytes, unsigned int dstY, void* srcHost, unsigned int srcPitch, unsigned int srcXInBytes, unsigned int srcY, unsigned int widthInBytes, unsigned int height) { CUDA_MEMCPY2D desc;@@ -283,11 +272,6 @@ CUresult CUDAAPI cuMemsetD32(CUdeviceptr dstDevice, unsigned int ui, size_t N) { return cuMemsetD32_v2(dstDevice, ui, N);-}--CUresult CUDAAPI cuTexRefSetAddress(size_t *ByteOffset, CUtexref hTexRef, CUdeviceptr dptr, size_t bytes)-{- return cuTexRefSetAddress_v2(ByteOffset, hTexRef, dptr, bytes); } #endif
cuda.cabal view
@@ -1,7 +1,7 @@ cabal-version: 1.24 Name: cuda-Version: 0.11.0.1+Version: 0.12.8.0 Synopsis: FFI binding to the CUDA interface for programming NVIDIA GPUs Description: The CUDA library provides a direct, general purpose C-like SPMD programming@@ -30,33 +30,20 @@ . * "Foreign.CUDA.Runtime" .- Tested with library versions up to CUDA-11.4. See also the+ Tested with library versions up to CUDA-12.8. See also the <https://travis-ci.org/tmcdonell/cuda travis-ci.org> build matrix for version compatibility. . [/NOTES:/] .- The setup script for this package requires at least Cabal-1.24. To upgrade,- execute one of:- .- * cabal users: @cabal install Cabal --constraint="Cabal >= 1.24"@- .- * stack users: @stack setup --upgrade-cabal@- .- Due to an interaction between GHC-8 and unified virtual address spaces in- CUDA, this package does not currently work with GHCi on ghc-8.0.1 (compiled- programs should work). See the following for more details:- .- * <https://github.com/tmcdonell/cuda/issues/39>- .- * <https://ghc.haskell.org/trac/ghc/ticket/12573>- .- The bug should be fixed in ghc-8.0.2 and beyond.- . For additional notes on installing on Windows, see: . * <https://github.com/tmcdonell/cuda/blob/master/WINDOWS.md> .+ This library is currently in __maintenance mode__. While we plan to release+ updates to keep the existing interface working with newer CUDA versions (as+ long as the underlying APIs remain available), no binding of new features is+ planned at the moment. Get in touch if you want to contribute. License: BSD3 License-file: LICENSE@@ -121,7 +108,6 @@ Foreign.CUDA.Driver.Module.Query Foreign.CUDA.Driver.Profiler Foreign.CUDA.Driver.Stream- Foreign.CUDA.Driver.Texture Foreign.CUDA.Driver.Unified Foreign.CUDA.Driver.Utils @@ -133,7 +119,6 @@ Foreign.CUDA.Runtime.Exec Foreign.CUDA.Runtime.Marshal Foreign.CUDA.Runtime.Stream- Foreign.CUDA.Runtime.Texture Foreign.CUDA.Runtime.Utils -- Extras@@ -151,6 +136,7 @@ build-depends: base >= 4.7 && < 5 , bytestring >= 0.10.4+ , containers , filepath >= 1.0 , template-haskell , uuid-types >= 1.0@@ -191,6 +177,6 @@ source-repository this type: git location: https://github.com/tmcdonell/cuda- tag: v0.11.0.1+ tag: v0.12.8.0 -- vim: nospell
src/Foreign/CUDA/Analysis/Device.chs view
@@ -19,8 +19,12 @@ #include "cbits/stubs.h" +import qualified Data.Set as Set+import Data.Set (Set) import Data.Int+import Data.IORef import Text.Show.Describe+import System.IO.Unsafe import Debug.Trace @@ -179,7 +183,17 @@ deviceResources :: DeviceProperties -> DeviceResources deviceResources = resources . computeCapability where- -- This is mostly extracted from tables in the CUDA occupancy calculator.+ -- Sources:+ -- [1] https://github.com/NVIDIA/cuda-samples/blob/7b60178984e96bc09d066077d5455df71fee2a9f/Common/helper_cuda.h+ -- - for: coresPerMP (line 643 _ConvertSMVer2Cores)+ -- - for: architecture names (line 695 _ConvertSMVer2ArchName)+ -- [2] https://docs.nvidia.com/cuda/cuda-c-programming-guide/index.html#features-and-technical-specifications-technical-specifications-per-compute-capability+ -- - for: maxGridsPerDevice+ -- - archived here: https://web.archive.org/web/20250409220108/https://docs.nvidia.com/cuda/cuda-c-programming-guide/index.html#features-and-technical-specifications-technical-specifications-per-compute-capability+ -- - reproduced here: https://en.wikipedia.org/w/index.php?title=CUDA&oldid=1285775690#Technical_specification (note: link to specific page version)+ -- [3] NVidia Nsight Compute+ -- - for: the other fields+ -- - left top "Start Activity" -> "Occupancy Calculator" -> "Launch"; tab "GPU Data" -- resources compute = case compute of Compute 1 0 -> resources (Compute 1 1) -- Tesla G80@@ -283,7 +297,7 @@ } Compute 5 2 -> (resources (Compute 5 0)) -- Maxwell GM20x { sharedMemPerMP = 98304- , maxRegPerBlock = 32768+ , maxRegPerBlock = 32768 -- value from [3], wrong in [2]? , warpAllocUnit = 2 } Compute 5 3 -> (resources (Compute 5 0)) -- Maxwell GM20B@@ -318,9 +332,15 @@ } Compute 6 2 -> (resources (Compute 6 0)) -- Pascal GP10B { coresPerMP = 128- , warpsPerMP = 128- , threadBlocksPerMP = 4096- , maxRegPerBlock = 32768+ -- Commit 4f75ea889c2ade2bd3eab377b51bb5bbd28bfbae changed warpsPerMP+ -- to 128, but [2] and [3] say 64 like CC 6.0; reverted back to 64 to+ -- match NVIDIA documentation.+ -- That commit also changed threadsPerMP (later mistakenly translated+ -- to threadBlocksPerMP in 9df19adec8efc9df761deab40cf04d27810d97d3)+ -- from 2048 to 4096, but again [2] and [3] retain 2048 so we keep it+ -- at that.+ , warpsPerMP = 64+ , maxRegPerBlock = 32768 -- value from [2], wrong in [3]? , warpAllocUnit = 4 , maxGridsPerDevice = 16 }@@ -346,7 +366,7 @@ Compute 7 2 -> (resources (Compute 7 0)) -- Volta GV10B { maxGridsPerDevice = 16- , maxSharedMemPerBlock = 49152+ , maxSharedMemPerBlock = 49152 -- unsure why this is here; [2] and [3] say still 98304 } Compute 7 5 -> (resources (Compute 7 0)) -- Turing TU1xx@@ -376,15 +396,92 @@ , warpRegAllocUnit = 256 , maxGridsPerDevice = 128 }- Compute 8 6 -> (resources (Compute 8 0)) -- Ampere GA102- { warpsPerMP = 48+ { coresPerMP = 128+ , warpsPerMP = 48 , threadsPerMP = 1536 , threadBlocksPerMP = 16 , sharedMemPerMP = 102400 , maxSharedMemPerBlock = 102400 }+ Compute 8 7 -> (resources (Compute 8 0)) -- Ampere+ { coresPerMP = 128+ , warpsPerMP = 48+ , threadsPerMP = 1536+ , threadBlocksPerMP = 16+ }+ Compute 8 9 -> (resources (Compute 8 0)) -- Ada+ { coresPerMP = 128+ , warpsPerMP = 48+ , threadsPerMP = 1536+ , threadBlocksPerMP = 24+ , sharedMemPerMP = 102400+ , maxSharedMemPerBlock = 102400+ } + Compute 9 0 -> DeviceResources -- Hopper+ { threadsPerWarp = 32+ , coresPerMP = 128+ , warpsPerMP = 64+ , threadsPerMP = 2048+ , threadBlocksPerMP = 32+ , sharedMemPerMP = 233472+ , maxSharedMemPerBlock = 233472+ , regFileSizePerMP = 65536+ , maxRegPerBlock = 65536+ , regAllocUnit = 256+ , regAllocationStyle = Warp+ , maxRegPerThread = 255+ , sharedMemAllocUnit = 128+ , warpAllocUnit = 4+ , warpRegAllocUnit = 256+ , maxGridsPerDevice = 128+ }++ Compute 10 0 -> DeviceResources -- Blackwell+ { threadsPerWarp = 32+ , coresPerMP = 128+ , warpsPerMP = 64+ , threadsPerMP = 2048+ , threadBlocksPerMP = 32+ , sharedMemPerMP = 233472+ , maxSharedMemPerBlock = 233472+ , regFileSizePerMP = 65536+ , maxRegPerBlock = 65536+ , regAllocUnit = 256+ , regAllocationStyle = Warp+ , maxRegPerThread = 255+ , sharedMemAllocUnit = 128+ , warpAllocUnit = 4+ , warpRegAllocUnit = 256+ , maxGridsPerDevice = 128+ }+ Compute 10 1 -> (resources (Compute 10 0)) -- Blackwell+ { warpsPerMP = 48+ , threadsPerMP = 1536+ , threadBlocksPerMP = 24+ }++ Compute 12 0 -> DeviceResources -- Blackwell+ { threadsPerWarp = 32+ , coresPerMP = 128+ , warpsPerMP = 48+ , threadsPerMP = 1536+ , threadBlocksPerMP = 24+ , sharedMemPerMP = 102400+ , maxSharedMemPerBlock = 102400+ , regFileSizePerMP = 65536+ , maxRegPerBlock = 65536+ , regAllocUnit = 256+ , regAllocationStyle = Warp+ , maxRegPerThread = 255+ , sharedMemAllocUnit = 128+ , warpAllocUnit = 4+ , warpRegAllocUnit = 256+ , maxGridsPerDevice = 128+ }++ -- Something might have gone wrong, or the library just needs to be -- updated for the next generation of hardware, in which case we just want -- to pick a sensible default and carry on.@@ -393,7 +490,30 @@ -- However, it should be OK because all library functions run in IO, so it -- is likely the user code is as well. --- _ -> trace warning $ resources (Compute 6 0)- where warning = unlines [ "*** Warning: Unknown CUDA device compute capability: " ++ show compute- , "*** Please submit a bug report at https://github.com/tmcdonell/cuda/issues" ]+ _ -> case warningForCC compute of+ Just warning -> trace warning defaultResources+ Nothing -> defaultResources + defaultResources = resources (Compute 6 0)++ -- All this logic is to ensure the warning is only shown once per unknown+ -- compute capability. This sounds not worth it, but in practice, it is:+ -- empirically, an unknown compute capability often leads to /screenfuls/+ -- of warnings in accelerate-llvm-ptx otherwise.+ {-# NOINLINE warningForCC #-}+ warningForCC :: Compute -> Maybe String+ warningForCC compute = unsafePerformIO $ do+ unseen <- atomicModifyIORef' warningShown $ \seen ->+ -- This is just one tree traversal; lookup-insert would be two traversals.+ let seen' = Set.insert compute seen+ in (seen', Set.size seen' > Set.size seen)+ return $ if unseen+ then Just $ unlines+ [ "*** Warning: Unknown CUDA device compute capability: " ++ show compute+ , "*** Please submit a bug report at https://github.com/tmcdonell/cuda/issues"+ , "*** (This warning will only be shown once for this compute capability)" ]+ else Nothing++ {-# NOINLINE warningShown #-}+ warningShown :: IORef (Set Compute)+ warningShown = unsafePerformIO $ newIORef mempty
src/Foreign/CUDA/Analysis/Occupancy.hs view
@@ -28,6 +28,14 @@ -- the number in the @.cubin@ file to the amount you dynamically allocate at run -- time to get the correct shared memory usage. --+-- __Warning__: Like the official Occupancy Calculator in NVidia Nsight+-- Compute, the calculator in this module does not support or consider Thread+-- Block Clusters+-- (<https://docs.nvidia.com/cuda/cuda-c-programming-guide/#thread-block-clusters>)+-- that have been introduced with compute capability 9.0 (Hopper). If you use+-- thread block clusters in your kernels, the results you get with the+-- functions in this module may not be accurate. Profile and measure.+-- -- /Notes About Occupancy/ -- -- Higher occupancy does not necessarily mean higher performance. If a kernel
src/Foreign/CUDA/Driver/Graph/Capture.chs view
@@ -152,11 +152,21 @@ #if CUDA_VERSION < 10010 info :: Stream -> IO (Status, Int64) info = requireSDK 'info 10.1-#else+#elif CUDA_VERSION < 12000 {# fun unsafe cuStreamGetCaptureInfo as info { useStream `Stream' , alloca- `Status' peekEnum* , alloca- `Int64' peekIntConv*+ }+ -> `()' checkStatus*- #}+#else+{# fun unsafe cuStreamGetCaptureInfo_v2 as info+ { useStream `Stream'+ , alloca- `Status' peekEnum*+ , alloca- `Int64' peekIntConv*+ , alloca- `Graph'+ , alloca- `Node'+ , alloca- `CSize' } -> `()' checkStatus*- #} #endif
src/Foreign/CUDA/Driver/Module/Query.chs view
@@ -16,7 +16,7 @@ module Foreign.CUDA.Driver.Module.Query ( -- ** Querying module inhabitants- getFun, getPtr, getTex,+ getFun, getPtr, ) where @@ -28,7 +28,6 @@ import Foreign.CUDA.Driver.Exec import Foreign.CUDA.Driver.Marshal ( peekDeviceHandle ) import Foreign.CUDA.Driver.Module.Base-import Foreign.CUDA.Driver.Texture import Foreign.CUDA.Internal.C2HS import Foreign.CUDA.Ptr @@ -86,26 +85,6 @@ {# fun unsafe cuModuleGetGlobal { alloca- `DevicePtr a' peekDeviceHandle* , alloca- `Int' peekIntConv*- , useModule `Module'- , useAsCString* `ShortByteString'- }- -> `Status' cToEnum #}----- |--- Return a handle to a texture reference. This texture reference handle--- should not be destroyed, as the texture will be destroyed automatically--- when the module is unloaded.------ <http://docs.nvidia.com/cuda/cuda-driver-api/group__CUDA__MODULE.html#group__CUDA__MODULE_1g9607dcbf911c16420d5264273f2b5608>----{-# INLINEABLE getTex #-}-getTex :: Module -> ShortByteString -> IO Texture-getTex !mdl !name = resultIfFound "texture" name =<< cuModuleGetTexRef mdl name--{-# INLINE cuModuleGetTexRef #-}-{# fun unsafe cuModuleGetTexRef- { alloca- `Texture' peekTex* , useModule `Module' , useAsCString* `ShortByteString' }
src/Foreign/CUDA/Driver/Stream.chs view
@@ -334,6 +334,7 @@ write32 ptr val stream flags = nothingIfOk =<< cuStreamWriteValue32 stream ptr val flags {-# INLINE cuStreamWriteValue32 #-}+#if CUDA_VERSION < 12000 {# fun unsafe cuStreamWriteValue32 { useStream `Stream' , useDeviceHandle `DevicePtr Word32'@@ -341,7 +342,16 @@ , combineBitMasks `[StreamWriteFlag]' } -> `Status' cToEnum #}+#else+{# fun unsafe cuStreamWriteValue32_v2 as cuStreamWriteValue32+ { useStream `Stream'+ , useDeviceHandle `DevicePtr Word32'+ , `Word32'+ , combineBitMasks `[StreamWriteFlag]'+ }+ -> `Status' cToEnum #} #endif+#endif {-# INLINE write64 #-} write64 :: DevicePtr Word64 -> Word64 -> Stream -> [StreamWriteFlag] -> IO ()@@ -351,6 +361,7 @@ write64 ptr val stream flags = nothingIfOk =<< cuStreamWriteValue64 stream ptr val flags {-# INLINE cuStreamWriteValue64 #-}+#if CUDA_VERSION < 12000 {# fun unsafe cuStreamWriteValue64 { useStream `Stream' , useDeviceHandle `DevicePtr Word64'@@ -358,7 +369,16 @@ , combineBitMasks `[StreamWriteFlag]' } -> `Status' cToEnum #}+#else+{# fun unsafe cuStreamWriteValue64_v2 as cuStreamWriteValue64+ { useStream `Stream'+ , useDeviceHandle `DevicePtr Word64'+ , `Word64'+ , combineBitMasks `[StreamWriteFlag]'+ }+ -> `Status' cToEnum #} #endif+#endif -- | Wait on a memory location. Work ordered after the operation will block@@ -388,13 +408,22 @@ wait32 ptr val stream flags = nothingIfOk =<< cuStreamWaitValue32 stream ptr val flags {-# INLINE cuStreamWaitValue32 #-}+#if CUDA_VERSION < 12000 {# fun unsafe cuStreamWaitValue32 { useStream `Stream' , useDeviceHandle `DevicePtr Word32' , `Word32' , combineBitMasks `[StreamWaitFlag]' } -> `Status' cToEnum #}+#else+{# fun unsafe cuStreamWaitValue32_v2 as cuStreamWaitValue32+ { useStream `Stream'+ , useDeviceHandle `DevicePtr Word32'+ , `Word32'+ , combineBitMasks `[StreamWaitFlag]'+ } -> `Status' cToEnum #} #endif+#endif {-# INLINE wait64 #-} wait64 :: DevicePtr Word64 -> Word64 -> Stream -> [StreamWaitFlag] -> IO ()@@ -404,12 +433,21 @@ wait64 ptr val stream flags = nothingIfOk =<< cuStreamWaitValue64 stream ptr val flags {-# INLINE cuStreamWaitValue64 #-}+#if CUDA_VERSION < 12000 {# fun unsafe cuStreamWaitValue64 { useStream `Stream' , useDeviceHandle `DevicePtr Word64' , `Word64' , combineBitMasks `[StreamWaitFlag]' } -> `Status' cToEnum #}+#else+{# fun unsafe cuStreamWaitValue64_v2 as cuStreamWaitValue64+ { useStream `Stream'+ , useDeviceHandle `DevicePtr Word64'+ , `Word64'+ , combineBitMasks `[StreamWaitFlag]'+ } -> `Status' cToEnum #}+#endif #endif
− src/Foreign/CUDA/Driver/Texture.chs
@@ -1,308 +0,0 @@-{-# LANGUAGE BangPatterns #-}-{-# LANGUAGE ForeignFunctionInterface #-}-{-# OPTIONS_HADDOCK prune #-}------------------------------------------------------------------------------------ |--- Module : Foreign.CUDA.Driver.Texture--- Copyright : [2009..2023] Trevor L. McDonell--- License : BSD------ Texture management for low-level driver interface--------------------------------------------------------------------------------------module Foreign.CUDA.Driver.Texture (-- -- * Texture Reference Management- Texture(..), Format(..), AddressMode(..), FilterMode(..), ReadMode(..),- bind, bind2D,- getAddressMode, getFilterMode, getFormat,- setAddressMode, setFilterMode, setFormat, setReadMode,-- -- Deprecated- create, destroy,-- -- Internal- peekTex--) where--#include "cbits/stubs.h"-{# context lib="cuda" #}---- Friends-import Foreign.CUDA.Ptr-import Foreign.CUDA.Driver.Error-import Foreign.CUDA.Driver.Marshal-import Foreign.CUDA.Internal.C2HS---- System-import Foreign-import Foreign.C-import Control.Monad--#if CUDA_VERSION >= 3020-{-# DEPRECATED create, destroy "as of CUDA version 3.2" #-}-#endif-------------------------------------------------------------------------------------- Data Types------------------------------------------------------------------------------------- |--- A texture reference----newtype Texture = Texture { useTexture :: {# type CUtexref #}}- deriving (Eq, Show)--instance Storable Texture where- sizeOf _ = sizeOf (undefined :: {# type CUtexref #})- alignment _ = alignment (undefined :: {# type CUtexref #})- peek p = Texture `fmap` peek (castPtr p)- poke p t = poke (castPtr p) (useTexture t)---- |--- Texture reference addressing modes----{# enum CUaddress_mode as AddressMode- { underscoreToCase }- with prefix="CU_TR_ADDRESS_MODE" deriving (Eq, Show) #}---- |--- Texture reference filtering mode----{# enum CUfilter_mode as FilterMode- { underscoreToCase }- with prefix="CU_TR_FILTER_MODE" deriving (Eq, Show) #}---- |--- Texture read mode options----#c-typedef enum CUtexture_flag_enum {- CU_TEXTURE_FLAG_READ_AS_INTEGER = CU_TRSF_READ_AS_INTEGER,- CU_TEXTURE_FLAG_NORMALIZED_COORDINATES = CU_TRSF_NORMALIZED_COORDINATES,- CU_TEXTURE_FLAG_SRGB = CU_TRSF_SRGB-} CUtexture_flag;-#endc--{# enum CUtexture_flag as ReadMode- { underscoreToCase- , CU_TEXTURE_FLAG_SRGB as SRGB }- with prefix="CU_TEXTURE_FLAG" deriving (Eq, Show) #}---- |--- Texture data formats----{# enum CUarray_format as Format- { underscoreToCase- , UNSIGNED_INT8 as Word8- , UNSIGNED_INT16 as Word16- , UNSIGNED_INT32 as Word32- , SIGNED_INT8 as Int8- , SIGNED_INT16 as Int16- , SIGNED_INT32 as Int32 }- with prefix="CU_AD_FORMAT" deriving (Eq, Show) #}-------------------------------------------------------------------------------------- Texture management------------------------------------------------------------------------------------- |--- Create a new texture reference. Once created, the application must call--- 'setPtr' to associate the reference with allocated memory. Other texture--- reference functions are used to specify the format and interpretation to be--- used when the memory is read through this reference.------ <http://docs.nvidia.com/cuda/cuda-driver-api/group__CUDA__TEXREF__DEPRECATED.html#group__CUDA__TEXREF__DEPRECATED_1g0084fabe2c6d28ffcf9d9f5c7164f16c>----{-# INLINEABLE create #-}-create :: IO Texture-create = resultIfOk =<< cuTexRefCreate--{-# INLINE cuTexRefCreate #-}-{# fun unsafe cuTexRefCreate- { alloca- `Texture' peekTex* } -> `Status' cToEnum #}----- |--- Destroy a texture reference.------ <http://docs.nvidia.com/cuda/cuda-driver-api/group__CUDA__TEXREF__DEPRECATED.html#group__CUDA__TEXREF__DEPRECATED_1gea8edbd6cf9f97e6ab2b41fc6785519d>----{-# INLINEABLE destroy #-}-destroy :: Texture -> IO ()-destroy !tex = nothingIfOk =<< cuTexRefDestroy tex--{-# INLINE cuTexRefDestroy #-}-{# fun unsafe cuTexRefDestroy- { useTexture `Texture' } -> `Status' cToEnum #}----- |--- Bind a linear array address of the given size (bytes) as a texture--- reference. Any previously bound references are unbound.------ <http://docs.nvidia.com/cuda/cuda-driver-api/group__CUDA__TEXREF.html#group__CUDA__TEXREF_1g44ef7e5055192d52b3d43456602b50a8>----{-# INLINEABLE bind #-}-bind :: Texture -> DevicePtr a -> Int64 -> IO ()-bind !tex !dptr !bytes = nothingIfOk =<< cuTexRefSetAddress tex dptr bytes--{-# INLINE cuTexRefSetAddress #-}-{# fun unsafe cuTexRefSetAddress- { alloca- `Int'- , useTexture `Texture'- , useDeviceHandle `DevicePtr a'- , `Int64' } -> `Status' cToEnum #}----- |--- Bind a linear address range to the given texture reference as a--- two-dimensional arena. Any previously bound reference is unbound. Note that--- calls to 'setFormat' can not follow a call to 'bind2D' for the same texture--- reference.------ <http://docs.nvidia.com/cuda/cuda-driver-api/group__CUDA__TEXREF.html#group__CUDA__TEXREF_1g26f709bbe10516681913d1ffe8756ee2>----{-# INLINEABLE bind2D #-}-bind2D :: Texture -> Format -> Int -> DevicePtr a -> (Int,Int) -> Int64 -> IO ()-bind2D !tex !fmt !chn !dptr (!width,!height) !pitch =- nothingIfOk =<< cuTexRefSetAddress2D_simple tex fmt chn dptr width height pitch--{-# INLINE cuTexRefSetAddress2D_simple #-}-{# fun unsafe cuTexRefSetAddress2D_simple- { useTexture `Texture'- , cFromEnum `Format'- , `Int'- , useDeviceHandle `DevicePtr a'- , `Int'- , `Int'- , `Int64' } -> `Status' cToEnum #}----- |--- Get the addressing mode used by a texture reference, corresponding to the--- given dimension (currently the only supported dimension values are 0 or 1).------ <http://docs.nvidia.com/cuda/cuda-driver-api/group__CUDA__TEXREF.html#group__CUDA__TEXREF_1gfb367d93dc1d20aab0cf8ce70d543b33>----{-# INLINEABLE getAddressMode #-}-getAddressMode :: Texture -> Int -> IO AddressMode-getAddressMode !tex !dim = resultIfOk =<< cuTexRefGetAddressMode tex dim--{-# INLINE cuTexRefGetAddressMode #-}-{# fun unsafe cuTexRefGetAddressMode- { alloca- `AddressMode' peekEnum*- , useTexture `Texture'- , `Int' } -> `Status' cToEnum #}----- |--- Get the filtering mode used by a texture reference.------ <http://docs.nvidia.com/cuda/cuda-driver-api/group__CUDA__TEXREF.html#group__CUDA__TEXREF_1g2439e069746f69b940f2f4dbc78cdf87>----{-# INLINEABLE getFilterMode #-}-getFilterMode :: Texture -> IO FilterMode-getFilterMode !tex = resultIfOk =<< cuTexRefGetFilterMode tex--{-# INLINE cuTexRefGetFilterMode #-}-{# fun unsafe cuTexRefGetFilterMode- { alloca- `FilterMode' peekEnum*- , useTexture `Texture' } -> `Status' cToEnum #}----- |--- Get the data format and number of channel components of the bound texture.------ <http://docs.nvidia.com/cuda/cuda-driver-api/group__CUDA__TEXREF.html#group__CUDA__TEXREF_1g90936eb6c7c4434a609e1160c278ae53>----{-# INLINEABLE getFormat #-}-getFormat :: Texture -> IO (Format, Int)-getFormat !tex = do- (!status,!fmt,!dim) <- cuTexRefGetFormat tex- resultIfOk (status,(fmt,dim))--{-# INLINE cuTexRefGetFormat #-}-{# fun unsafe cuTexRefGetFormat- { alloca- `Format' peekEnum*- , alloca- `Int' peekIntConv*- , useTexture `Texture' } -> `Status' cToEnum #}----- |--- Specify the addressing mode for the given dimension of a texture reference.------ <http://docs.nvidia.com/cuda/cuda-driver-api/group__CUDA__TEXREF.html#group__CUDA__TEXREF_1g85f4a13eeb94c8072f61091489349bcb>----{-# INLINEABLE setAddressMode #-}-setAddressMode :: Texture -> Int -> AddressMode -> IO ()-setAddressMode !tex !dim !mode = nothingIfOk =<< cuTexRefSetAddressMode tex dim mode--{-# INLINE cuTexRefSetAddressMode #-}-{# fun unsafe cuTexRefSetAddressMode- { useTexture `Texture'- , `Int'- , cFromEnum `AddressMode' } -> `Status' cToEnum #}----- |--- Specify the filtering mode to be used when reading memory through a texture--- reference.------ <http://docs.nvidia.com/cuda/cuda-driver-api/group__CUDA__TEXREF.html#group__CUDA__TEXREF_1g595d0af02c55576f8c835e4efd1f39c0>----{-# INLINEABLE setFilterMode #-}-setFilterMode :: Texture -> FilterMode -> IO ()-setFilterMode !tex !mode = nothingIfOk =<< cuTexRefSetFilterMode tex mode--{-# INLINE cuTexRefSetFilterMode #-}-{# fun unsafe cuTexRefSetFilterMode- { useTexture `Texture'- , cFromEnum `FilterMode' } -> `Status' cToEnum #}----- |--- Specify additional characteristics for reading and indexing the texture--- reference.------ <http://docs.nvidia.com/cuda/cuda-driver-api/group__CUDA__TEXREF.html#group__CUDA__TEXREF_1g554ffd896487533c36810f2e45bb7a28>----{-# INLINEABLE setReadMode #-}-setReadMode :: Texture -> ReadMode -> IO ()-setReadMode !tex !mode = nothingIfOk =<< cuTexRefSetFlags tex mode--{-# INLINE cuTexRefSetFlags #-}-{# fun unsafe cuTexRefSetFlags- { useTexture `Texture'- , cFromEnum `ReadMode' } -> `Status' cToEnum #}----- |--- Specify the format of the data and number of packed components per element to--- be read by the texture reference.------ <http://docs.nvidia.com/cuda/cuda-driver-api/group__CUDA__TEXREF.html#group__CUDA__TEXREF_1g05585ef8ea2fec728a03c6c8f87cf07a>----{-# INLINEABLE setFormat #-}-setFormat :: Texture -> Format -> Int -> IO ()-setFormat !tex !fmt !chn = nothingIfOk =<< cuTexRefSetFormat tex fmt chn--{-# INLINE cuTexRefSetFormat #-}-{# fun unsafe cuTexRefSetFormat- { useTexture `Texture'- , cFromEnum `Format'- , `Int' } -> `Status' cToEnum #}-------------------------------------------------------------------------------------- Internal-----------------------------------------------------------------------------------{-# INLINE peekTex #-}-peekTex :: Ptr {# type CUtexref #} -> IO Texture-peekTex = liftM Texture . peek-
src/Foreign/CUDA/Runtime/Device.chs view
@@ -219,9 +219,15 @@ props !n = resultIfOk =<< cudaGetDeviceProperties n {-# INLINE cudaGetDeviceProperties #-}+#if CUDA_VERSION < 12000 {# fun unsafe cudaGetDeviceProperties { alloca- `DeviceProperties' peek* , `Int' } -> `Status' cToEnum #}+#else+{# fun unsafe cudaGetDeviceProperties_v2 as cudaGetDeviceProperties+ { alloca- `DeviceProperties' peek*+ , `Int' } -> `Status' cToEnum #}+#endif -- |
− src/Foreign/CUDA/Runtime/Texture.chs
@@ -1,203 +0,0 @@-{-# LANGUAGE BangPatterns #-}-{-# LANGUAGE ForeignFunctionInterface #-}------------------------------------------------------------------------------------ |--- Module : Foreign.CUDA.Runtime.Texture--- Copyright : [2009..2023] Trevor L. McDonell--- License : BSD------ Texture references--------------------------------------------------------------------------------------module Foreign.CUDA.Runtime.Texture (-- -- * Texture Reference Management- Texture(..), FormatKind(..), AddressMode(..), FilterMode(..), FormatDesc(..),- bind, bind2D--) where---- Friends-import Foreign.CUDA.Ptr-import Foreign.CUDA.Runtime.Error-import Foreign.CUDA.Internal.C2HS---- System-import Data.Int-import Foreign-import Foreign.C--#include "cbits/stubs.h"-{# context lib="cudart" #}--#c-typedef struct textureReference textureReference;-typedef struct cudaChannelFormatDesc cudaChannelFormatDesc;-#endc------------------------------------------------------------------------------------- Data Types------------------------------------------------------------------------------------- |A texture reference----{# pointer *textureReference as ^ -> Texture #}--data Texture = Texture- {- normalised :: !Bool, -- ^ access texture using normalised coordinates [0.0,1.0)- filtering :: !FilterMode,- addressing :: !(AddressMode, AddressMode, AddressMode),- format :: !FormatDesc- }- deriving (Eq, Show)---- |Texture channel format kind----{# enum cudaChannelFormatKind as FormatKind- { }- with prefix="cudaChannelFormatKind" deriving (Eq, Show) #}---- |Texture addressing mode----{# enum cudaTextureAddressMode as AddressMode- { }- with prefix="cudaAddressMode" deriving (Eq, Show) #}---- |Texture filtering mode----{# enum cudaTextureFilterMode as FilterMode- { }- with prefix="cudaFilterMode" deriving (Eq, Show) #}----- |A description of how memory read through the texture cache should be--- interpreted, including the kind of data and the number of bits of each--- component (x,y,z and w, respectively).----{# pointer *cudaChannelFormatDesc as ^ foreign -> FormatDesc nocode #}--data FormatDesc = FormatDesc- {- depth :: !(Int,Int,Int,Int),- kind :: !FormatKind- }- deriving (Eq, Show)--instance Storable FormatDesc where- sizeOf _ = {# sizeof cudaChannelFormatDesc #}- alignment _ = alignment (undefined :: Ptr ())-- peek p = do- dx <- cIntConv `fmap` {# get cudaChannelFormatDesc.x #} p- dy <- cIntConv `fmap` {# get cudaChannelFormatDesc.y #} p- dz <- cIntConv `fmap` {# get cudaChannelFormatDesc.z #} p- dw <- cIntConv `fmap` {# get cudaChannelFormatDesc.w #} p- df <- cToEnum `fmap` {# get cudaChannelFormatDesc.f #} p- return $ FormatDesc (dx,dy,dz,dw) df-- poke p (FormatDesc (x,y,z,w) k) = do- {# set cudaChannelFormatDesc.x #} p (cIntConv x)- {# set cudaChannelFormatDesc.y #} p (cIntConv y)- {# set cudaChannelFormatDesc.z #} p (cIntConv z)- {# set cudaChannelFormatDesc.w #} p (cIntConv w)- {# set cudaChannelFormatDesc.f #} p (cFromEnum k)---instance Storable Texture where- sizeOf _ = {# sizeof textureReference #}- alignment _ = alignment (undefined :: Ptr ())-- peek p = do- norm <- cToBool `fmap` {# get textureReference.normalized #} p- fmt <- cToEnum `fmap` {# get textureReference.filterMode #} p- dsc <- peek . castPtr =<< {# get textureReference.channelDesc #} p- [x,y,z] <- peekArrayWith cToEnum 3 =<< {# get textureReference.addressMode #} p- return $ Texture norm fmt (x,y,z) dsc-- poke p (Texture norm fmt (x,y,z) dsc) = do- {# set textureReference.normalized #} p (cFromBool norm)- {# set textureReference.filterMode #} p (cFromEnum fmt)- withArray (map cFromEnum [x,y,z]) ({# set textureReference.addressMode #} p)-- -- c2hs is returning the wrong type for structs-within-structs- dscptr <- {# get textureReference.channelDesc #} p- poke (castPtr dscptr) dsc-------------------------------------------------------------------------------------- Texture References------------------------------------------------------------------------------------- |Bind the memory area associated with the device pointer to a texture--- reference given by the named symbol. Any previously bound references are--- unbound.----{-# INLINEABLE bind #-}-bind :: String -> Texture -> DevicePtr a -> Int64 -> IO ()-bind !name !tex !dptr !bytes = do- ref <- getTex name- poke ref tex- nothingIfOk =<< cudaBindTexture ref dptr (format tex) bytes--{-# INLINE cudaBindTexture #-}-{# fun unsafe cudaBindTexture- { alloca- `Int'- , id `TextureReference'- , dptr `DevicePtr a'- , with_* `FormatDesc'- , `Int64' } -> `Status' cToEnum #}- where dptr = useDevicePtr . castDevPtr---- |Bind the two-dimensional memory area to the texture reference associated--- with the given symbol. The size of the area is constrained by (width,height)--- in texel units, and the row pitch in bytes. Any previously bound references--- are unbound.----{-# INLINEABLE bind2D #-}-bind2D :: String -> Texture -> DevicePtr a -> (Int,Int) -> Int64 -> IO ()-bind2D !name !tex !dptr (!width,!height) !bytes = do- ref <- getTex name- poke ref tex- nothingIfOk =<< cudaBindTexture2D ref dptr (format tex) width height bytes--{-# INLINE cudaBindTexture2D #-}-{# fun unsafe cudaBindTexture2D- { alloca- `Int'- , id `TextureReference'- , dptr `DevicePtr a'- , with_* `FormatDesc'- , `Int'- , `Int'- , `Int64' } -> `Status' cToEnum #}- where dptr = useDevicePtr . castDevPtr----- |Returns the texture reference associated with the given symbol----{-# INLINEABLE getTex #-}-getTex :: String -> IO TextureReference-getTex !name = resultIfOk =<< cudaGetTextureReference name--{-# INLINE cudaGetTextureReference #-}-{# fun unsafe cudaGetTextureReference- { alloca- `Ptr Texture' peek*- , withCString_* `String' } -> `Status' cToEnum #}-------------------------------------------------------------------------------------- Internal-----------------------------------------------------------------------------------{-# INLINE with_ #-}-with_ :: Storable a => a -> (Ptr a -> IO b) -> IO b-with_ = with----- CUDA 5.0 changed the types of some attributes from char* to void*----{-# INLINE withCString_ #-}-withCString_ :: String -> (Ptr a -> IO b) -> IO b-withCString_ !str !fn = withCString str (fn . castPtr)-