packages feed

cuda 0.11.0.1 → 0.12.8.0

raw patch · 13 files changed

+282/−609 lines, 13 filesdep +containerssetup-changednew-uploaderPVP ok

version bump matches the API change (PVP)

Dependencies added: containers

API changes (from Hackage documentation)

- Foreign.CUDA.Driver.Device: CanUse64BitStreamMemOpsV2 :: DeviceAttribute
- Foreign.CUDA.Driver.Device: CanUseStreamMemOps :: DeviceAttribute
- Foreign.CUDA.Driver.Device: CanUseStreamWaitValueNorV2 :: DeviceAttribute
- Foreign.CUDA.Driver.Module: Compute20 :: JITTarget
- Foreign.CUDA.Driver.Module: Compute21 :: JITTarget
- Foreign.CUDA.Driver.Module.Base: Compute20 :: JITTarget
- Foreign.CUDA.Driver.Module.Base: Compute21 :: JITTarget
- Foreign.CUDA.Driver.Module.Base: loadFile :: FilePath -> IO Module
- Foreign.CUDA.Driver.Module.Query: getTex :: Module -> ShortByteString -> IO Texture
- Foreign.CUDA.Driver.Texture: Bc1Unorm :: Format
- Foreign.CUDA.Driver.Texture: Bc1UnormSrgb :: Format
- Foreign.CUDA.Driver.Texture: Bc2Unorm :: Format
- Foreign.CUDA.Driver.Texture: Bc2UnormSrgb :: Format
- Foreign.CUDA.Driver.Texture: Bc3Unorm :: Format
- Foreign.CUDA.Driver.Texture: Bc3UnormSrgb :: Format
- Foreign.CUDA.Driver.Texture: Bc4Snorm :: Format
- Foreign.CUDA.Driver.Texture: Bc4Unorm :: Format
- Foreign.CUDA.Driver.Texture: Bc5Snorm :: Format
- Foreign.CUDA.Driver.Texture: Bc5Unorm :: Format
- Foreign.CUDA.Driver.Texture: Bc6hSf16 :: Format
- Foreign.CUDA.Driver.Texture: Bc6hUf16 :: Format
- Foreign.CUDA.Driver.Texture: Bc7Unorm :: Format
- Foreign.CUDA.Driver.Texture: Bc7UnormSrgb :: Format
- Foreign.CUDA.Driver.Texture: Border :: AddressMode
- Foreign.CUDA.Driver.Texture: Clamp :: AddressMode
- Foreign.CUDA.Driver.Texture: Float :: Format
- Foreign.CUDA.Driver.Texture: Half :: Format
- Foreign.CUDA.Driver.Texture: Int16 :: Format
- Foreign.CUDA.Driver.Texture: Int32 :: Format
- Foreign.CUDA.Driver.Texture: Int8 :: Format
- Foreign.CUDA.Driver.Texture: Linear :: FilterMode
- Foreign.CUDA.Driver.Texture: Mirror :: AddressMode
- Foreign.CUDA.Driver.Texture: NormalizedCoordinates :: ReadMode
- Foreign.CUDA.Driver.Texture: Nv12 :: Format
- Foreign.CUDA.Driver.Texture: Point :: FilterMode
- Foreign.CUDA.Driver.Texture: ReadAsInteger :: ReadMode
- Foreign.CUDA.Driver.Texture: SRGB :: ReadMode
- Foreign.CUDA.Driver.Texture: SnormInt16x1 :: Format
- Foreign.CUDA.Driver.Texture: SnormInt16x2 :: Format
- Foreign.CUDA.Driver.Texture: SnormInt16x4 :: Format
- Foreign.CUDA.Driver.Texture: SnormInt8x1 :: Format
- Foreign.CUDA.Driver.Texture: SnormInt8x2 :: Format
- Foreign.CUDA.Driver.Texture: SnormInt8x4 :: Format
- Foreign.CUDA.Driver.Texture: Texture :: Ptr () -> Texture
- Foreign.CUDA.Driver.Texture: UnormInt16x1 :: Format
- Foreign.CUDA.Driver.Texture: UnormInt16x2 :: Format
- Foreign.CUDA.Driver.Texture: UnormInt16x4 :: Format
- Foreign.CUDA.Driver.Texture: UnormInt8x1 :: Format
- Foreign.CUDA.Driver.Texture: UnormInt8x2 :: Format
- Foreign.CUDA.Driver.Texture: UnormInt8x4 :: Format
- Foreign.CUDA.Driver.Texture: Word16 :: Format
- Foreign.CUDA.Driver.Texture: Word32 :: Format
- Foreign.CUDA.Driver.Texture: Word8 :: Format
- Foreign.CUDA.Driver.Texture: Wrap :: AddressMode
- Foreign.CUDA.Driver.Texture: [useTexture] :: Texture -> Ptr ()
- Foreign.CUDA.Driver.Texture: bind :: Texture -> DevicePtr a -> Int64 -> IO ()
- Foreign.CUDA.Driver.Texture: bind2D :: Texture -> Format -> Int -> DevicePtr a -> (Int, Int) -> Int64 -> IO ()
- Foreign.CUDA.Driver.Texture: create :: IO Texture
- Foreign.CUDA.Driver.Texture: data AddressMode
- Foreign.CUDA.Driver.Texture: data FilterMode
- Foreign.CUDA.Driver.Texture: data Format
- Foreign.CUDA.Driver.Texture: data ReadMode
- Foreign.CUDA.Driver.Texture: getAddressMode :: Texture -> Int -> IO AddressMode
- Foreign.CUDA.Driver.Texture: getFilterMode :: Texture -> IO FilterMode
- Foreign.CUDA.Driver.Texture: getFormat :: Texture -> IO (Format, Int)
- Foreign.CUDA.Driver.Texture: instance Foreign.Storable.Storable Foreign.CUDA.Driver.Texture.Texture
- Foreign.CUDA.Driver.Texture: instance GHC.Classes.Eq Foreign.CUDA.Driver.Texture.AddressMode
- Foreign.CUDA.Driver.Texture: instance GHC.Classes.Eq Foreign.CUDA.Driver.Texture.FilterMode
- Foreign.CUDA.Driver.Texture: instance GHC.Classes.Eq Foreign.CUDA.Driver.Texture.Format
- Foreign.CUDA.Driver.Texture: instance GHC.Classes.Eq Foreign.CUDA.Driver.Texture.ReadMode
- Foreign.CUDA.Driver.Texture: instance GHC.Classes.Eq Foreign.CUDA.Driver.Texture.Texture
- Foreign.CUDA.Driver.Texture: instance GHC.Enum.Enum Foreign.CUDA.Driver.Texture.AddressMode
- Foreign.CUDA.Driver.Texture: instance GHC.Enum.Enum Foreign.CUDA.Driver.Texture.FilterMode
- Foreign.CUDA.Driver.Texture: instance GHC.Enum.Enum Foreign.CUDA.Driver.Texture.Format
- Foreign.CUDA.Driver.Texture: instance GHC.Enum.Enum Foreign.CUDA.Driver.Texture.ReadMode
- Foreign.CUDA.Driver.Texture: instance GHC.Show.Show Foreign.CUDA.Driver.Texture.AddressMode
- Foreign.CUDA.Driver.Texture: instance GHC.Show.Show Foreign.CUDA.Driver.Texture.FilterMode
- Foreign.CUDA.Driver.Texture: instance GHC.Show.Show Foreign.CUDA.Driver.Texture.Format
- Foreign.CUDA.Driver.Texture: instance GHC.Show.Show Foreign.CUDA.Driver.Texture.ReadMode
- Foreign.CUDA.Driver.Texture: instance GHC.Show.Show Foreign.CUDA.Driver.Texture.Texture
- Foreign.CUDA.Driver.Texture: newtype Texture
- Foreign.CUDA.Driver.Texture: setAddressMode :: Texture -> Int -> AddressMode -> IO ()
- Foreign.CUDA.Driver.Texture: setFilterMode :: Texture -> FilterMode -> IO ()
- Foreign.CUDA.Driver.Texture: setFormat :: Texture -> Format -> Int -> IO ()
- Foreign.CUDA.Driver.Texture: setReadMode :: Texture -> ReadMode -> IO ()
- Foreign.CUDA.Runtime.Texture: Border :: AddressMode
- Foreign.CUDA.Runtime.Texture: Clamp :: AddressMode
- Foreign.CUDA.Runtime.Texture: Float :: FormatKind
- Foreign.CUDA.Runtime.Texture: FormatDesc :: !(Int, Int, Int, Int) -> !FormatKind -> FormatDesc
- Foreign.CUDA.Runtime.Texture: Linear :: FilterMode
- Foreign.CUDA.Runtime.Texture: Mirror :: AddressMode
- Foreign.CUDA.Runtime.Texture: NV12 :: FormatKind
- Foreign.CUDA.Runtime.Texture: None :: FormatKind
- Foreign.CUDA.Runtime.Texture: Point :: FilterMode
- Foreign.CUDA.Runtime.Texture: Signed :: FormatKind
- Foreign.CUDA.Runtime.Texture: SignedBlockCompressed4 :: FormatKind
- Foreign.CUDA.Runtime.Texture: SignedBlockCompressed5 :: FormatKind
- Foreign.CUDA.Runtime.Texture: SignedBlockCompressed6H :: FormatKind
- Foreign.CUDA.Runtime.Texture: SignedNormalized16X1 :: FormatKind
- Foreign.CUDA.Runtime.Texture: SignedNormalized16X2 :: FormatKind
- Foreign.CUDA.Runtime.Texture: SignedNormalized16X4 :: FormatKind
- Foreign.CUDA.Runtime.Texture: SignedNormalized8X1 :: FormatKind
- Foreign.CUDA.Runtime.Texture: SignedNormalized8X2 :: FormatKind
- Foreign.CUDA.Runtime.Texture: SignedNormalized8X4 :: FormatKind
- Foreign.CUDA.Runtime.Texture: Texture :: !Bool -> !FilterMode -> !(AddressMode, AddressMode, AddressMode) -> !FormatDesc -> Texture
- Foreign.CUDA.Runtime.Texture: Unsigned :: FormatKind
- Foreign.CUDA.Runtime.Texture: UnsignedBlockCompressed1 :: FormatKind
- Foreign.CUDA.Runtime.Texture: UnsignedBlockCompressed1SRGB :: FormatKind
- Foreign.CUDA.Runtime.Texture: UnsignedBlockCompressed2 :: FormatKind
- Foreign.CUDA.Runtime.Texture: UnsignedBlockCompressed2SRGB :: FormatKind
- Foreign.CUDA.Runtime.Texture: UnsignedBlockCompressed3 :: FormatKind
- Foreign.CUDA.Runtime.Texture: UnsignedBlockCompressed3SRGB :: FormatKind
- Foreign.CUDA.Runtime.Texture: UnsignedBlockCompressed4 :: FormatKind
- Foreign.CUDA.Runtime.Texture: UnsignedBlockCompressed5 :: FormatKind
- Foreign.CUDA.Runtime.Texture: UnsignedBlockCompressed6H :: FormatKind
- Foreign.CUDA.Runtime.Texture: UnsignedBlockCompressed7 :: FormatKind
- Foreign.CUDA.Runtime.Texture: UnsignedBlockCompressed7SRGB :: FormatKind
- Foreign.CUDA.Runtime.Texture: UnsignedNormalized16X1 :: FormatKind
- Foreign.CUDA.Runtime.Texture: UnsignedNormalized16X2 :: FormatKind
- Foreign.CUDA.Runtime.Texture: UnsignedNormalized16X4 :: FormatKind
- Foreign.CUDA.Runtime.Texture: UnsignedNormalized8X1 :: FormatKind
- Foreign.CUDA.Runtime.Texture: UnsignedNormalized8X2 :: FormatKind
- Foreign.CUDA.Runtime.Texture: UnsignedNormalized8X4 :: FormatKind
- Foreign.CUDA.Runtime.Texture: Wrap :: AddressMode
- Foreign.CUDA.Runtime.Texture: [addressing] :: Texture -> !(AddressMode, AddressMode, AddressMode)
- Foreign.CUDA.Runtime.Texture: [depth] :: FormatDesc -> !(Int, Int, Int, Int)
- Foreign.CUDA.Runtime.Texture: [filtering] :: Texture -> !FilterMode
- Foreign.CUDA.Runtime.Texture: [format] :: Texture -> !FormatDesc
- Foreign.CUDA.Runtime.Texture: [kind] :: FormatDesc -> !FormatKind
- Foreign.CUDA.Runtime.Texture: [normalised] :: Texture -> !Bool
- Foreign.CUDA.Runtime.Texture: bind :: String -> Texture -> DevicePtr a -> Int64 -> IO ()
- Foreign.CUDA.Runtime.Texture: bind2D :: String -> Texture -> DevicePtr a -> (Int, Int) -> Int64 -> IO ()
- Foreign.CUDA.Runtime.Texture: data AddressMode
- Foreign.CUDA.Runtime.Texture: data FilterMode
- Foreign.CUDA.Runtime.Texture: data FormatDesc
- Foreign.CUDA.Runtime.Texture: data FormatKind
- Foreign.CUDA.Runtime.Texture: data Texture
- Foreign.CUDA.Runtime.Texture: instance Foreign.Storable.Storable Foreign.CUDA.Runtime.Texture.FormatDesc
- Foreign.CUDA.Runtime.Texture: instance Foreign.Storable.Storable Foreign.CUDA.Runtime.Texture.Texture
- Foreign.CUDA.Runtime.Texture: instance GHC.Classes.Eq Foreign.CUDA.Runtime.Texture.AddressMode
- Foreign.CUDA.Runtime.Texture: instance GHC.Classes.Eq Foreign.CUDA.Runtime.Texture.FilterMode
- Foreign.CUDA.Runtime.Texture: instance GHC.Classes.Eq Foreign.CUDA.Runtime.Texture.FormatDesc
- Foreign.CUDA.Runtime.Texture: instance GHC.Classes.Eq Foreign.CUDA.Runtime.Texture.FormatKind
- Foreign.CUDA.Runtime.Texture: instance GHC.Classes.Eq Foreign.CUDA.Runtime.Texture.Texture
- Foreign.CUDA.Runtime.Texture: instance GHC.Enum.Enum Foreign.CUDA.Runtime.Texture.AddressMode
- Foreign.CUDA.Runtime.Texture: instance GHC.Enum.Enum Foreign.CUDA.Runtime.Texture.FilterMode
- Foreign.CUDA.Runtime.Texture: instance GHC.Enum.Enum Foreign.CUDA.Runtime.Texture.FormatKind
- Foreign.CUDA.Runtime.Texture: instance GHC.Show.Show Foreign.CUDA.Runtime.Texture.AddressMode
- Foreign.CUDA.Runtime.Texture: instance GHC.Show.Show Foreign.CUDA.Runtime.Texture.FilterMode
- Foreign.CUDA.Runtime.Texture: instance GHC.Show.Show Foreign.CUDA.Runtime.Texture.FormatDesc
- Foreign.CUDA.Runtime.Texture: instance GHC.Show.Show Foreign.CUDA.Runtime.Texture.FormatKind
- Foreign.CUDA.Runtime.Texture: instance GHC.Show.Show Foreign.CUDA.Runtime.Texture.Texture
+ Foreign.CUDA.Driver: CigEnabled :: Limit
+ Foreign.CUDA.Driver: CigShmemFallbackEnabled :: Limit
+ Foreign.CUDA.Driver: CoredumpEnable :: ContextFlag
+ Foreign.CUDA.Driver: OnlyPartialNativeAtomicSupported :: PeerAttribute
+ Foreign.CUDA.Driver: ShmemSize :: Limit
+ Foreign.CUDA.Driver: SyncMemops :: ContextFlag
+ Foreign.CUDA.Driver: UserCoredumpEnable :: ContextFlag
+ Foreign.CUDA.Driver.Context.Base: CoredumpEnable :: ContextFlag
+ Foreign.CUDA.Driver.Context.Base: SyncMemops :: ContextFlag
+ Foreign.CUDA.Driver.Context.Base: UserCoredumpEnable :: ContextFlag
+ Foreign.CUDA.Driver.Context.Config: CigEnabled :: Limit
+ Foreign.CUDA.Driver.Context.Config: CigShmemFallbackEnabled :: Limit
+ Foreign.CUDA.Driver.Context.Config: ShmemSize :: Limit
+ Foreign.CUDA.Driver.Context.Peer: OnlyPartialNativeAtomicSupported :: PeerAttribute
+ Foreign.CUDA.Driver.Device: CanUse64BitStreamMemOpsV1 :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: CanUseStreamMemOpsV1 :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: CanUseStreamWaitValueNorV1 :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: D3d12CigSupported :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: GpuPciDeviceId :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: GpuPciSubsystemId :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: HandleTypeFabricSupported :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: HostNumaId :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: HostNumaMemoryPoolsSupported :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: HostNumaMultinodeIpcSupported :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: HostNumaVirtualMemoryManagementSupported :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: IpcEventSupported :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: MemDecompressAlgorithmMask :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: MemDecompressMaximumLength :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: MemSyncDomainCount :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: MpsEnabled :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: MulticastSupported :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: NumaConfig :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: NumaId :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: TensorMapAccessSupported :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: UnifiedFunctionPointers :: DeviceAttribute
+ Foreign.CUDA.Driver.Device: VulkanCigSupported :: DeviceAttribute
+ Foreign.CUDA.Driver.Error: CdpNotSupported :: Status
+ Foreign.CUDA.Driver.Error: CdpVersionMismatch :: Status
+ Foreign.CUDA.Driver.Error: Contained :: Status
+ Foreign.CUDA.Driver.Error: FunctionNotLoaded :: Status
+ Foreign.CUDA.Driver.Error: InvalidResourceConfiguration :: Status
+ Foreign.CUDA.Driver.Error: InvalidResourceType :: Status
+ Foreign.CUDA.Driver.Error: KeyRotation :: Status
+ Foreign.CUDA.Driver.Error: LossyQuery :: Status
+ Foreign.CUDA.Driver.Error: TensorMemoryLeak :: Status
+ Foreign.CUDA.Driver.Error: UnsupportedDevsideSync :: Status
+ Foreign.CUDA.Driver.Graph.Base: Conditional :: NodeType
+ Foreign.CUDA.Driver.Graph.Build: Conditional :: NodeType
+ Foreign.CUDA.Driver.Module: Compute100 :: JITTarget
+ Foreign.CUDA.Driver.Module: Compute100a :: JITTarget
+ Foreign.CUDA.Driver.Module: Compute100f :: JITTarget
+ Foreign.CUDA.Driver.Module: Compute101 :: JITTarget
+ Foreign.CUDA.Driver.Module: Compute101a :: JITTarget
+ Foreign.CUDA.Driver.Module: Compute101f :: JITTarget
+ Foreign.CUDA.Driver.Module: Compute103 :: JITTarget
+ Foreign.CUDA.Driver.Module: Compute103a :: JITTarget
+ Foreign.CUDA.Driver.Module: Compute103f :: JITTarget
+ Foreign.CUDA.Driver.Module: Compute120 :: JITTarget
+ Foreign.CUDA.Driver.Module: Compute120a :: JITTarget
+ Foreign.CUDA.Driver.Module: Compute120f :: JITTarget
+ Foreign.CUDA.Driver.Module: Compute121 :: JITTarget
+ Foreign.CUDA.Driver.Module: Compute121a :: JITTarget
+ Foreign.CUDA.Driver.Module: Compute121f :: JITTarget
+ Foreign.CUDA.Driver.Module: Compute90a :: JITTarget
+ Foreign.CUDA.Driver.Module.Base: Compute100 :: JITTarget
+ Foreign.CUDA.Driver.Module.Base: Compute100a :: JITTarget
+ Foreign.CUDA.Driver.Module.Base: Compute100f :: JITTarget
+ Foreign.CUDA.Driver.Module.Base: Compute101 :: JITTarget
+ Foreign.CUDA.Driver.Module.Base: Compute101a :: JITTarget
+ Foreign.CUDA.Driver.Module.Base: Compute101f :: JITTarget
+ Foreign.CUDA.Driver.Module.Base: Compute103 :: JITTarget
+ Foreign.CUDA.Driver.Module.Base: Compute103a :: JITTarget
+ Foreign.CUDA.Driver.Module.Base: Compute103f :: JITTarget
+ Foreign.CUDA.Driver.Module.Base: Compute120 :: JITTarget
+ Foreign.CUDA.Driver.Module.Base: Compute120a :: JITTarget
+ Foreign.CUDA.Driver.Module.Base: Compute120f :: JITTarget
+ Foreign.CUDA.Driver.Module.Base: Compute121 :: JITTarget
+ Foreign.CUDA.Driver.Module.Base: Compute121a :: JITTarget
+ Foreign.CUDA.Driver.Module.Base: Compute121f :: JITTarget
+ Foreign.CUDA.Driver.Module.Base: Compute90a :: JITTarget
+ Foreign.CUDA.Driver.Module.Base: loadData :: ByteString -> IO Module
+ Foreign.CUDA.Runtime.Error: CdpNotSupported :: Status
+ Foreign.CUDA.Runtime.Error: CdpVersionMismatch :: Status
+ Foreign.CUDA.Runtime.Error: Contained :: Status
+ Foreign.CUDA.Runtime.Error: FunctionNotLoaded :: Status
+ Foreign.CUDA.Runtime.Error: InvalidResourceConfiguration :: Status
+ Foreign.CUDA.Runtime.Error: InvalidResourceType :: Status
+ Foreign.CUDA.Runtime.Error: LossyQuery :: Status
+ Foreign.CUDA.Runtime.Error: TensorMemoryLeak :: Status
+ Foreign.CUDA.Runtime.Error: UnsupportedDevSideSync :: Status

Files

CHANGELOG.md view
@@ -10,6 +10,19 @@ because NVIDIA are A-OK introducing breaking changes in minor updates.  +## [0.12.8.0] - ???+### Added+  * Support for CUDA-12+      - Thanks to @noahmartinwilliams on GitHub for helping out!++### Removed+  * The following modules have been deprecated for a long time, and have+    finally been removed in CUDA-12:+      - `Foreign.CUDA.Driver.Texture`+      - `Foreign.CUDA.Runtime.Texture`+    Support for Texture Objects (their replacement) is missing in these+    bindings so far. Contributions welcome.+ ## [0.11.0.1] - 2023-08-15 ### Fixed   * Build fixes for GHC 9.2 .. 9.6
README.md view
@@ -25,8 +25,13 @@  ## Missing functionality -An incomplete list of missing bindings. Pull requests welcome!+_This library is currently in **maintenance mode**. While we plan to release+updates to keep the existing interface working with newer CUDA versions (as+long as the underlying APIs remain available), no binding of new features is+planned at the moment. Get in touch if you want to contribute._ +Here is an incomplete historical list of missing bindings. Pull requests welcome!+ ### CUDA-9  - cuLaunchCooperativeKernelMultiDevice@@ -145,3 +150,23 @@ - cuGraphMemAllocNodeGetParams - cuGraphMemFreeNodeGetParams +### CUDA-12++A lot. PRs welcome.+++# Old compatibility notes++The setup script for this package requires at least Cabal-1.24. If you run into trouble with this:++* Cabal users: ensure you are using a new `cabal` executable and have run `cabal update` anywhere in the last few years. If you have previously run `cabal install` on libraries and have a broken environment as a result, remove `~/.ghc/<platfom>/environments/default`.+* Stack users: one may attempt @stack setup --upgrade-cabal@.++Due to an interaction between GHC-8 and unified virtual address spaces in+CUDA, this package does not currently work with GHCi on ghc-8.0.1 (compiled+programs should work). See the following for more details:++* <https://github.com/tmcdonell/cuda/issues/39>+* <https://ghc.haskell.org/trac/ghc/ticket/12573>++The bug should be fixed in ghc-8.0.2 and beyond.
Setup.hs view
@@ -36,6 +36,7 @@  import Control.Exception import Control.Monad+import Data.Char (isDigit) import Data.Function import Data.List import Data.Maybe@@ -140,7 +141,7 @@     -> IO HookedBuildInfo libraryBuildInfo verbosity profile installPath platform@(Platform arch os) ghcVersion extraLibs extraIncludes = do   let-      libraryPaths      = cudaLibraryPath platform installPath : extraLibs+      libraryPaths      = cudaLibraryPaths platform installPath ++ extraLibs       includePaths      = cudaIncludePath platform installPath : extraIncludes        takeFirstExisting paths = do@@ -215,18 +216,19 @@ cudaIncludePath _ installPath = installPath </> "include"  --- Return the location of the libraries relative to the base CUDA installation.+-- Return the potential locations of the libraries relative to the base CUDA installation. ---cudaLibraryPath :: Platform -> FilePath -> FilePath-cudaLibraryPath (Platform arch os) installPath = installPath </> libpath+cudaLibraryPaths :: Platform -> FilePath -> [FilePath]+cudaLibraryPaths (Platform arch os) installPath = [ installPath </> path | path <- libpaths ]   where-    libpath =+    libpaths =       case (os, arch) of-        (Windows, I386)   -> "lib/Win32"-        (Windows, X86_64) -> "lib/x64"-        (OSX,     _)      -> "lib"    -- MacOS does not distinguish 32- vs. 64-bit paths-        (_,       X86_64) -> "lib64"  -- treat all others similarly-        _                 -> "lib"+        (Windows, I386)    -> ["lib/Win32"]+        (Windows, X86_64)  -> ["lib/x64"]+        (OSX,     _)       -> ["lib"]    -- MacOS does not distinguish 32- vs. 64-bit paths+        (_,       X86_64)  -> ["lib64", "lib"]  -- prefer lib64 for 64-bit systems+        (_,       AArch64) -> ["lib64", "lib"]+        _                  -> ["lib"]           -- otherwise   -- On Windows and OSX we use different libraries depending on whether we are@@ -264,7 +266,9 @@     -> [FilePath]     -> IO [FilePath] cudaGhciLibrariesWindows platform installPath libraries = do-  candidates <- mapM (importLibraryToDLLFileName platform) [ cudaLibraryPath platform installPath </> lib <.> "lib" | lib <- libraries ]+  candidates <- mapM (importLibraryToDLLFileName platform)+                     [ libPath </> lib <.> "lib" | libPath <- cudaLibraryPaths platform installPath+                                                 , lib <- libraries ]   return [ dropExtension dll | Just dll <- candidates ]  @@ -460,7 +464,7 @@     -> Platform     -> IO FilePath findCUDAInstallPath verbosity platform = do-  result <- findFirstValidLocation verbosity platform (candidateCUDAInstallPaths verbosity platform)+  result <- findFirstValidLocation verbosity platform =<< candidateCUDAInstallPaths verbosity platform   case result of     Just installPath -> do       notice verbosity $ printf "Found CUDA toolkit at: %s (set CUDA_PATH to override this)" installPath@@ -547,19 +551,15 @@ candidateCUDAInstallPaths     :: Verbosity     -> Platform-    -> [(IO FilePath, String)]-candidateCUDAInstallPaths verbosity platform =-  [ (getEnv "CUDA_PATH",      "environment variable CUDA_PATH")-  , (findInPath,              "nvcc compiler executable in PATH")-  , (return defaultPath,      printf "default install location (%s)" defaultPath)-  , (getEnv "CUDA_PATH_V9_1", "environment variable CUDA_PATH_V9_1")-  , (getEnv "CUDA_PATH_V9_0", "environment variable CUDA_PATH_V9_0")-  , (getEnv "CUDA_PATH_V8_0", "environment variable CUDA_PATH_V8_0")-  , (getEnv "CUDA_PATH_V7_5", "environment variable CUDA_PATH_V7_5")-  , (getEnv "CUDA_PATH_V7_0", "environment variable CUDA_PATH_V7_0")-  , (getEnv "CUDA_PATH_V6_5", "environment variable CUDA_PATH_V6_5")-  , (getEnv "CUDA_PATH_V6_0", "environment variable CUDA_PATH_V6_0")-  ]+    -> IO [(IO FilePath, String)]+candidateCUDAInstallPaths verbosity platform = do+  let defaults =+        [ (getEnv "CUDA_PATH", "environment variable CUDA_PATH")+        , (findInPath,         "nvcc compiler executable in PATH")+        , (return defaultPath, printf "default install location (%s)" defaultPath)+        ]+  verVars <- versionedVars+  return $ defaults ++ verVars   where     findInPath :: IO FilePath     findInPath = do@@ -570,6 +570,21 @@      defaultPath :: FilePath     defaultPath = defaultCUDAInstallPath platform++    versionedVars :: IO [(IO FilePath, String)]+    versionedVars = do+      pairs <- getEnvironment+      let sorted = sort (mapMaybe (\(k, v) -> (,k,v) <$> parseCudaPathVerVar k) pairs)+      return [(return v, "environment variable " ++ k) | (_, k, v) <- sorted]++    parseCudaPathVerVar :: String -> Maybe (Int, Int)+    parseCudaPathVerVar var+      | ("CUDA_PATH_V", s1) <- splitAt 11 var+      , (n1, '_':n2) <- span isDigit s1, not (null n1)+      , all isDigit n2, not (null n2)+      = Just (read n1, read n2)+      | otherwise+      = Nothing   -- NOTE: this function throws an exception when there is no `nvcc` in PATH.
cbits/stubs.c view
@@ -22,17 +22,6 @@ } #endif -CUresult cuTexRefSetAddress2D_simple(CUtexref tex, CUarray_format format, unsigned int numChannels, CUdeviceptr dptr, size_t width, size_t height, size_t pitch)-{-    CUDA_ARRAY_DESCRIPTOR desc;-    desc.Format      = format;-    desc.NumChannels = numChannels;-    desc.Width       = width;-    desc.Height      = height;--    return cuTexRefSetAddress2D(tex, &desc, dptr, pitch);-}- CUresult cuMemcpy2DHtoD(CUdeviceptr dstDevice, unsigned int dstPitch, unsigned int dstXInBytes, unsigned int dstY, void* srcHost, unsigned int srcPitch, unsigned int srcXInBytes, unsigned int srcY, unsigned int widthInBytes, unsigned int height) {     CUDA_MEMCPY2D desc;@@ -283,11 +272,6 @@ CUresult CUDAAPI cuMemsetD32(CUdeviceptr dstDevice, unsigned int ui, size_t N) {     return cuMemsetD32_v2(dstDevice, ui, N);-}--CUresult CUDAAPI cuTexRefSetAddress(size_t *ByteOffset, CUtexref hTexRef, CUdeviceptr dptr, size_t bytes)-{-    return cuTexRefSetAddress_v2(ByteOffset, hTexRef, dptr, bytes); } #endif 
cuda.cabal view
@@ -1,7 +1,7 @@ cabal-version:          1.24  Name:                   cuda-Version:                0.11.0.1+Version:                0.12.8.0 Synopsis:               FFI binding to the CUDA interface for programming NVIDIA GPUs Description:     The CUDA library provides a direct, general purpose C-like SPMD programming@@ -30,33 +30,20 @@     .     * "Foreign.CUDA.Runtime"     .-    Tested with library versions up to CUDA-11.4. See also the+    Tested with library versions up to CUDA-12.8. See also the     <https://travis-ci.org/tmcdonell/cuda travis-ci.org> build matrix for     version compatibility.     .     [/NOTES:/]     .-    The setup script for this package requires at least Cabal-1.24. To upgrade,-    execute one of:-    .-    * cabal users: @cabal install Cabal --constraint="Cabal >= 1.24"@-    .-    * stack users: @stack setup --upgrade-cabal@-    .-    Due to an interaction between GHC-8 and unified virtual address spaces in-    CUDA, this package does not currently work with GHCi on ghc-8.0.1 (compiled-    programs should work). See the following for more details:-    .-    * <https://github.com/tmcdonell/cuda/issues/39>-    .-    * <https://ghc.haskell.org/trac/ghc/ticket/12573>-    .-    The bug should be fixed in ghc-8.0.2 and beyond.-    .     For additional notes on installing on Windows, see:     .     * <https://github.com/tmcdonell/cuda/blob/master/WINDOWS.md>     .+    This library is currently in __maintenance mode__. While we plan to release+    updates to keep the existing interface working with newer CUDA versions (as+    long as the underlying APIs remain available), no binding of new features is+    planned at the moment. Get in touch if you want to contribute.  License:                BSD3 License-file:           LICENSE@@ -121,7 +108,6 @@       Foreign.CUDA.Driver.Module.Query       Foreign.CUDA.Driver.Profiler       Foreign.CUDA.Driver.Stream-      Foreign.CUDA.Driver.Texture       Foreign.CUDA.Driver.Unified       Foreign.CUDA.Driver.Utils @@ -133,7 +119,6 @@       Foreign.CUDA.Runtime.Exec       Foreign.CUDA.Runtime.Marshal       Foreign.CUDA.Runtime.Stream-      Foreign.CUDA.Runtime.Texture       Foreign.CUDA.Runtime.Utils        -- Extras@@ -151,6 +136,7 @@   build-depends:       base              >= 4.7 && < 5     , bytestring        >= 0.10.4+    , containers     , filepath          >= 1.0     , template-haskell     , uuid-types        >= 1.0@@ -191,6 +177,6 @@ source-repository this     type:               git     location:           https://github.com/tmcdonell/cuda-    tag:                v0.11.0.1+    tag:                v0.12.8.0  -- vim: nospell
src/Foreign/CUDA/Analysis/Device.chs view
@@ -19,8 +19,12 @@  #include "cbits/stubs.h" +import qualified Data.Set as Set+import Data.Set (Set) import Data.Int+import Data.IORef import Text.Show.Describe+import System.IO.Unsafe  import Debug.Trace @@ -179,7 +183,17 @@ deviceResources :: DeviceProperties -> DeviceResources deviceResources = resources . computeCapability   where-    -- This is mostly extracted from tables in the CUDA occupancy calculator.+    -- Sources:+    -- [1] https://github.com/NVIDIA/cuda-samples/blob/7b60178984e96bc09d066077d5455df71fee2a9f/Common/helper_cuda.h+    --    - for: coresPerMP (line 643 _ConvertSMVer2Cores)+    --    - for: architecture names (line 695 _ConvertSMVer2ArchName)+    -- [2] https://docs.nvidia.com/cuda/cuda-c-programming-guide/index.html#features-and-technical-specifications-technical-specifications-per-compute-capability+    --    - for: maxGridsPerDevice+    --    - archived here: https://web.archive.org/web/20250409220108/https://docs.nvidia.com/cuda/cuda-c-programming-guide/index.html#features-and-technical-specifications-technical-specifications-per-compute-capability+    --    - reproduced here: https://en.wikipedia.org/w/index.php?title=CUDA&oldid=1285775690#Technical_specification (note: link to specific page version)+    -- [3] NVidia Nsight Compute+    --    - for: the other fields+    --    - left top "Start Activity" -> "Occupancy Calculator" -> "Launch"; tab "GPU Data"     --     resources compute = case compute of       Compute 1 0 -> resources (Compute 1 1)      -- Tesla G80@@ -283,7 +297,7 @@         }       Compute 5 2 -> (resources (Compute 5 0))    -- Maxwell GM20x         { sharedMemPerMP        = 98304-        , maxRegPerBlock        = 32768+        , maxRegPerBlock        = 32768  -- value from [3], wrong in [2]?         , warpAllocUnit         = 2         }       Compute 5 3 -> (resources (Compute 5 0))    -- Maxwell GM20B@@ -318,9 +332,15 @@         }       Compute 6 2 -> (resources (Compute 6 0))    -- Pascal GP10B         { coresPerMP            = 128-        , warpsPerMP            = 128-        , threadBlocksPerMP     = 4096-        , maxRegPerBlock        = 32768+        -- Commit 4f75ea889c2ade2bd3eab377b51bb5bbd28bfbae changed warpsPerMP+        -- to 128, but [2] and [3] say 64 like CC 6.0; reverted back to 64 to+        -- match NVIDIA documentation.+        -- That commit also changed threadsPerMP (later mistakenly translated+        -- to threadBlocksPerMP in 9df19adec8efc9df761deab40cf04d27810d97d3)+        -- from 2048 to 4096, but again [2] and [3] retain 2048 so we keep it+        -- at that.+        , warpsPerMP            = 64+        , maxRegPerBlock        = 32768  -- value from [2], wrong in [3]?         , warpAllocUnit         = 4         , maxGridsPerDevice     = 16         }@@ -346,7 +366,7 @@        Compute 7 2 -> (resources (Compute 7 0))    -- Volta GV10B         { maxGridsPerDevice     = 16-        , maxSharedMemPerBlock  = 49152+        , maxSharedMemPerBlock  = 49152  -- unsure why this is here; [2] and [3] say still 98304         }        Compute 7 5 -> (resources (Compute 7 0))    -- Turing TU1xx@@ -376,15 +396,92 @@         , warpRegAllocUnit      = 256         , maxGridsPerDevice     = 128         }-       Compute 8 6 -> (resources (Compute 8 0))    -- Ampere GA102-        { warpsPerMP            = 48+        { coresPerMP            = 128+        , warpsPerMP            = 48         , threadsPerMP          = 1536         , threadBlocksPerMP     = 16         , sharedMemPerMP        = 102400         , maxSharedMemPerBlock  = 102400         }+      Compute 8 7 -> (resources (Compute 8 0))    -- Ampere+        { coresPerMP            = 128+        , warpsPerMP            = 48+        , threadsPerMP          = 1536+        , threadBlocksPerMP     = 16+        }+      Compute 8 9 -> (resources (Compute 8 0))    -- Ada+        { coresPerMP            = 128+        , warpsPerMP            = 48+        , threadsPerMP          = 1536+        , threadBlocksPerMP     = 24+        , sharedMemPerMP        = 102400+        , maxSharedMemPerBlock  = 102400+        } +      Compute 9 0 -> DeviceResources              -- Hopper+        { threadsPerWarp        = 32+        , coresPerMP            = 128+        , warpsPerMP            = 64+        , threadsPerMP          = 2048+        , threadBlocksPerMP     = 32+        , sharedMemPerMP        = 233472+        , maxSharedMemPerBlock  = 233472+        , regFileSizePerMP      = 65536+        , maxRegPerBlock        = 65536+        , regAllocUnit          = 256+        , regAllocationStyle    = Warp+        , maxRegPerThread       = 255+        , sharedMemAllocUnit    = 128+        , warpAllocUnit         = 4+        , warpRegAllocUnit      = 256+        , maxGridsPerDevice     = 128+        }++      Compute 10 0 -> DeviceResources             -- Blackwell+        { threadsPerWarp        = 32+        , coresPerMP            = 128+        , warpsPerMP            = 64+        , threadsPerMP          = 2048+        , threadBlocksPerMP     = 32+        , sharedMemPerMP        = 233472+        , maxSharedMemPerBlock  = 233472+        , regFileSizePerMP      = 65536+        , maxRegPerBlock        = 65536+        , regAllocUnit          = 256+        , regAllocationStyle    = Warp+        , maxRegPerThread       = 255+        , sharedMemAllocUnit    = 128+        , warpAllocUnit         = 4+        , warpRegAllocUnit      = 256+        , maxGridsPerDevice     = 128+        }+      Compute 10 1 -> (resources (Compute 10 0))  -- Blackwell+        { warpsPerMP            = 48+        , threadsPerMP          = 1536+        , threadBlocksPerMP     = 24+        }++      Compute 12 0 -> DeviceResources             -- Blackwell+        { threadsPerWarp        = 32+        , coresPerMP            = 128+        , warpsPerMP            = 48+        , threadsPerMP          = 1536+        , threadBlocksPerMP     = 24+        , sharedMemPerMP        = 102400+        , maxSharedMemPerBlock  = 102400+        , regFileSizePerMP      = 65536+        , maxRegPerBlock        = 65536+        , regAllocUnit          = 256+        , regAllocationStyle    = Warp+        , maxRegPerThread       = 255+        , sharedMemAllocUnit    = 128+        , warpAllocUnit         = 4+        , warpRegAllocUnit      = 256+        , maxGridsPerDevice     = 128+        }++       -- Something might have gone wrong, or the library just needs to be       -- updated for the next generation of hardware, in which case we just want       -- to pick a sensible default and carry on.@@ -393,7 +490,30 @@       -- However, it should be OK because all library functions run in IO, so it       -- is likely the user code is as well.       ---      _           -> trace warning $ resources (Compute 6 0)-        where warning = unlines [ "*** Warning: Unknown CUDA device compute capability: " ++ show compute-                                , "*** Please submit a bug report at https://github.com/tmcdonell/cuda/issues" ]+      _ -> case warningForCC compute of+             Just warning -> trace warning defaultResources+             Nothing      -> defaultResources +    defaultResources = resources (Compute 6 0)++    -- All this logic is to ensure the warning is only shown once per unknown+    -- compute capability. This sounds not worth it, but in practice, it is:+    -- empirically, an unknown compute capability often leads to /screenfuls/+    -- of warnings in accelerate-llvm-ptx otherwise.+    {-# NOINLINE warningForCC #-}+    warningForCC :: Compute -> Maybe String+    warningForCC compute = unsafePerformIO $ do+      unseen <- atomicModifyIORef' warningShown $ \seen ->+                  -- This is just one tree traversal; lookup-insert would be two traversals.+                  let seen' = Set.insert compute seen+                  in (seen', Set.size seen' > Set.size seen)+      return $ if unseen+        then Just $ unlines+               [ "*** Warning: Unknown CUDA device compute capability: " ++ show compute+               , "*** Please submit a bug report at https://github.com/tmcdonell/cuda/issues"+               , "*** (This warning will only be shown once for this compute capability)" ]+        else Nothing++    {-# NOINLINE warningShown #-}+    warningShown :: IORef (Set Compute)+    warningShown = unsafePerformIO $ newIORef mempty
src/Foreign/CUDA/Analysis/Occupancy.hs view
@@ -28,6 +28,14 @@ -- the number in the @.cubin@ file to the amount you dynamically allocate at run -- time to get the correct shared memory usage. --+-- __Warning__: Like the official Occupancy Calculator in NVidia Nsight+-- Compute, the calculator in this module does not support or consider Thread+-- Block Clusters+-- (<https://docs.nvidia.com/cuda/cuda-c-programming-guide/#thread-block-clusters>)+-- that have been introduced with compute capability 9.0 (Hopper). If you use+-- thread block clusters in your kernels, the results you get with the+-- functions in this module may not be accurate. Profile and measure.+-- -- /Notes About Occupancy/ -- -- Higher occupancy does not necessarily mean higher performance.  If a kernel
src/Foreign/CUDA/Driver/Graph/Capture.chs view
@@ -152,11 +152,21 @@ #if CUDA_VERSION < 10010 info :: Stream -> IO (Status, Int64) info = requireSDK 'info 10.1-#else+#elif CUDA_VERSION < 12000 {# fun unsafe cuStreamGetCaptureInfo as info   { useStream `Stream'   , alloca-   `Status' peekEnum*   , alloca-   `Int64'  peekIntConv*+  }+  -> `()' checkStatus*- #}+#else+{# fun unsafe cuStreamGetCaptureInfo_v2 as info+  { useStream `Stream'+  , alloca-   `Status' peekEnum*+  , alloca-   `Int64'  peekIntConv*+  , alloca-   `Graph'+  , alloca-   `Node'+  , alloca-   `CSize'   }   -> `()' checkStatus*- #} #endif
src/Foreign/CUDA/Driver/Module/Query.chs view
@@ -16,7 +16,7 @@ module Foreign.CUDA.Driver.Module.Query (    -- ** Querying module inhabitants-  getFun, getPtr, getTex,+  getFun, getPtr,  ) where @@ -28,7 +28,6 @@ import Foreign.CUDA.Driver.Exec import Foreign.CUDA.Driver.Marshal                      ( peekDeviceHandle ) import Foreign.CUDA.Driver.Module.Base-import Foreign.CUDA.Driver.Texture import Foreign.CUDA.Internal.C2HS import Foreign.CUDA.Ptr @@ -86,26 +85,6 @@ {# fun unsafe cuModuleGetGlobal   { alloca-       `DevicePtr a'     peekDeviceHandle*   , alloca-       `Int'             peekIntConv*-  , useModule     `Module'-  , useAsCString* `ShortByteString'-  }-  -> `Status' cToEnum #}----- |--- Return a handle to a texture reference. This texture reference handle--- should not be destroyed, as the texture will be destroyed automatically--- when the module is unloaded.------ <http://docs.nvidia.com/cuda/cuda-driver-api/group__CUDA__MODULE.html#group__CUDA__MODULE_1g9607dcbf911c16420d5264273f2b5608>----{-# INLINEABLE getTex #-}-getTex :: Module -> ShortByteString -> IO Texture-getTex !mdl !name = resultIfFound "texture" name =<< cuModuleGetTexRef mdl name--{-# INLINE cuModuleGetTexRef #-}-{# fun unsafe cuModuleGetTexRef-  { alloca-       `Texture'         peekTex*   , useModule     `Module'   , useAsCString* `ShortByteString'   }
src/Foreign/CUDA/Driver/Stream.chs view
@@ -334,6 +334,7 @@ write32 ptr val stream flags = nothingIfOk =<< cuStreamWriteValue32 stream ptr val flags  {-# INLINE cuStreamWriteValue32 #-}+#if CUDA_VERSION < 12000 {# fun unsafe cuStreamWriteValue32   { useStream       `Stream'   , useDeviceHandle `DevicePtr Word32'@@ -341,7 +342,16 @@   , combineBitMasks `[StreamWriteFlag]'   }   -> `Status' cToEnum #}+#else+{# fun unsafe cuStreamWriteValue32_v2 as cuStreamWriteValue32+  { useStream       `Stream'+  , useDeviceHandle `DevicePtr Word32'+  ,                 `Word32'+  , combineBitMasks `[StreamWriteFlag]'+  }+  -> `Status' cToEnum #} #endif+#endif  {-# INLINE write64 #-} write64 :: DevicePtr Word64 -> Word64 -> Stream -> [StreamWriteFlag] -> IO ()@@ -351,6 +361,7 @@ write64 ptr val stream flags = nothingIfOk =<< cuStreamWriteValue64 stream ptr val flags  {-# INLINE cuStreamWriteValue64 #-}+#if CUDA_VERSION < 12000 {# fun unsafe cuStreamWriteValue64   { useStream       `Stream'   , useDeviceHandle `DevicePtr Word64'@@ -358,7 +369,16 @@   , combineBitMasks `[StreamWriteFlag]'   }   -> `Status' cToEnum #}+#else+{# fun unsafe cuStreamWriteValue64_v2 as cuStreamWriteValue64+  { useStream       `Stream'+  , useDeviceHandle `DevicePtr Word64'+  ,                 `Word64'+  , combineBitMasks `[StreamWriteFlag]'+  }+  -> `Status' cToEnum #} #endif+#endif   -- | Wait on a memory location. Work ordered after the operation will block@@ -388,13 +408,22 @@ wait32 ptr val stream flags = nothingIfOk =<< cuStreamWaitValue32 stream ptr val flags  {-# INLINE cuStreamWaitValue32 #-}+#if CUDA_VERSION < 12000 {# fun unsafe cuStreamWaitValue32   { useStream       `Stream'   , useDeviceHandle `DevicePtr Word32'   ,                 `Word32'   , combineBitMasks `[StreamWaitFlag]'   } -> `Status' cToEnum #}+#else+{# fun unsafe cuStreamWaitValue32_v2 as cuStreamWaitValue32+  { useStream       `Stream'+  , useDeviceHandle `DevicePtr Word32'+  ,                 `Word32'+  , combineBitMasks `[StreamWaitFlag]'+  } -> `Status' cToEnum #} #endif+#endif  {-# INLINE wait64 #-} wait64 :: DevicePtr Word64 -> Word64 -> Stream -> [StreamWaitFlag] -> IO ()@@ -404,12 +433,21 @@ wait64 ptr val stream flags = nothingIfOk =<< cuStreamWaitValue64 stream ptr val flags  {-# INLINE cuStreamWaitValue64 #-}+#if CUDA_VERSION < 12000 {# fun unsafe cuStreamWaitValue64   { useStream       `Stream'   , useDeviceHandle `DevicePtr Word64'   ,                 `Word64'   , combineBitMasks `[StreamWaitFlag]'   } -> `Status' cToEnum #}+#else+{# fun unsafe cuStreamWaitValue64_v2 as cuStreamWaitValue64+  { useStream       `Stream'+  , useDeviceHandle `DevicePtr Word64'+  ,                 `Word64'+  , combineBitMasks `[StreamWaitFlag]'+  } -> `Status' cToEnum #}+#endif #endif  
− src/Foreign/CUDA/Driver/Texture.chs
@@ -1,308 +0,0 @@-{-# LANGUAGE BangPatterns             #-}-{-# LANGUAGE ForeignFunctionInterface #-}-{-# OPTIONS_HADDOCK prune #-}------------------------------------------------------------------------------------ |--- Module    : Foreign.CUDA.Driver.Texture--- Copyright : [2009..2023] Trevor L. McDonell--- License   : BSD------ Texture management for low-level driver interface--------------------------------------------------------------------------------------module Foreign.CUDA.Driver.Texture (--  -- * Texture Reference Management-  Texture(..), Format(..), AddressMode(..), FilterMode(..), ReadMode(..),-  bind, bind2D,-  getAddressMode, getFilterMode, getFormat,-  setAddressMode, setFilterMode, setFormat, setReadMode,--  -- Deprecated-  create, destroy,--  -- Internal-  peekTex--) where--#include "cbits/stubs.h"-{# context lib="cuda" #}---- Friends-import Foreign.CUDA.Ptr-import Foreign.CUDA.Driver.Error-import Foreign.CUDA.Driver.Marshal-import Foreign.CUDA.Internal.C2HS---- System-import Foreign-import Foreign.C-import Control.Monad--#if CUDA_VERSION >= 3020-{-# DEPRECATED create, destroy "as of CUDA version 3.2" #-}-#endif-------------------------------------------------------------------------------------- Data Types------------------------------------------------------------------------------------- |--- A texture reference----newtype Texture = Texture { useTexture :: {# type CUtexref #}}-  deriving (Eq, Show)--instance Storable Texture where-  sizeOf _    = sizeOf    (undefined :: {# type CUtexref #})-  alignment _ = alignment (undefined :: {# type CUtexref #})-  peek p      = Texture `fmap` peek (castPtr p)-  poke p t    = poke (castPtr p) (useTexture t)---- |--- Texture reference addressing modes----{# enum CUaddress_mode as AddressMode-  { underscoreToCase }-  with prefix="CU_TR_ADDRESS_MODE" deriving (Eq, Show) #}---- |--- Texture reference filtering mode----{# enum CUfilter_mode as FilterMode-  { underscoreToCase }-  with prefix="CU_TR_FILTER_MODE" deriving (Eq, Show) #}---- |--- Texture read mode options----#c-typedef enum CUtexture_flag_enum {-  CU_TEXTURE_FLAG_READ_AS_INTEGER        = CU_TRSF_READ_AS_INTEGER,-  CU_TEXTURE_FLAG_NORMALIZED_COORDINATES = CU_TRSF_NORMALIZED_COORDINATES,-  CU_TEXTURE_FLAG_SRGB                   = CU_TRSF_SRGB-} CUtexture_flag;-#endc--{# enum CUtexture_flag as ReadMode-  { underscoreToCase-  , CU_TEXTURE_FLAG_SRGB as SRGB }-  with prefix="CU_TEXTURE_FLAG" deriving (Eq, Show) #}---- |--- Texture data formats----{# enum CUarray_format as Format-  { underscoreToCase-  , UNSIGNED_INT8  as Word8-  , UNSIGNED_INT16 as Word16-  , UNSIGNED_INT32 as Word32-  , SIGNED_INT8    as Int8-  , SIGNED_INT16   as Int16-  , SIGNED_INT32   as Int32 }-  with prefix="CU_AD_FORMAT" deriving (Eq, Show) #}-------------------------------------------------------------------------------------- Texture management------------------------------------------------------------------------------------- |--- Create a new texture reference. Once created, the application must call--- 'setPtr' to associate the reference with allocated memory. Other texture--- reference functions are used to specify the format and interpretation to be--- used when the memory is read through this reference.------ <http://docs.nvidia.com/cuda/cuda-driver-api/group__CUDA__TEXREF__DEPRECATED.html#group__CUDA__TEXREF__DEPRECATED_1g0084fabe2c6d28ffcf9d9f5c7164f16c>----{-# INLINEABLE create #-}-create :: IO Texture-create = resultIfOk =<< cuTexRefCreate--{-# INLINE cuTexRefCreate #-}-{# fun unsafe cuTexRefCreate-  { alloca- `Texture' peekTex* } -> `Status' cToEnum #}----- |--- Destroy a texture reference.------ <http://docs.nvidia.com/cuda/cuda-driver-api/group__CUDA__TEXREF__DEPRECATED.html#group__CUDA__TEXREF__DEPRECATED_1gea8edbd6cf9f97e6ab2b41fc6785519d>----{-# INLINEABLE destroy #-}-destroy :: Texture -> IO ()-destroy !tex = nothingIfOk =<< cuTexRefDestroy tex--{-# INLINE cuTexRefDestroy #-}-{# fun unsafe cuTexRefDestroy-  { useTexture `Texture' } -> `Status' cToEnum #}----- |--- Bind a linear array address of the given size (bytes) as a texture--- reference. Any previously bound references are unbound.------ <http://docs.nvidia.com/cuda/cuda-driver-api/group__CUDA__TEXREF.html#group__CUDA__TEXREF_1g44ef7e5055192d52b3d43456602b50a8>----{-# INLINEABLE bind #-}-bind :: Texture -> DevicePtr a -> Int64 -> IO ()-bind !tex !dptr !bytes = nothingIfOk =<< cuTexRefSetAddress tex dptr bytes--{-# INLINE cuTexRefSetAddress #-}-{# fun unsafe cuTexRefSetAddress-  { alloca-         `Int'-  , useTexture      `Texture'-  , useDeviceHandle `DevicePtr a'-  ,                 `Int64'       } -> `Status' cToEnum #}----- |--- Bind a linear address range to the given texture reference as a--- two-dimensional arena. Any previously bound reference is unbound. Note that--- calls to 'setFormat' can not follow a call to 'bind2D' for the same texture--- reference.------ <http://docs.nvidia.com/cuda/cuda-driver-api/group__CUDA__TEXREF.html#group__CUDA__TEXREF_1g26f709bbe10516681913d1ffe8756ee2>----{-# INLINEABLE bind2D #-}-bind2D :: Texture -> Format -> Int -> DevicePtr a -> (Int,Int) -> Int64 -> IO ()-bind2D !tex !fmt !chn !dptr (!width,!height) !pitch =-  nothingIfOk =<< cuTexRefSetAddress2D_simple tex fmt chn dptr width height pitch--{-# INLINE cuTexRefSetAddress2D_simple #-}-{# fun unsafe cuTexRefSetAddress2D_simple-  { useTexture      `Texture'-  , cFromEnum       `Format'-  ,                 `Int'-  , useDeviceHandle `DevicePtr a'-  ,                 `Int'-  ,                 `Int'-  ,                 `Int64'       } -> `Status' cToEnum #}----- |--- Get the addressing mode used by a texture reference, corresponding to the--- given dimension (currently the only supported dimension values are 0 or 1).------ <http://docs.nvidia.com/cuda/cuda-driver-api/group__CUDA__TEXREF.html#group__CUDA__TEXREF_1gfb367d93dc1d20aab0cf8ce70d543b33>----{-# INLINEABLE getAddressMode #-}-getAddressMode :: Texture -> Int -> IO AddressMode-getAddressMode !tex !dim = resultIfOk =<< cuTexRefGetAddressMode tex dim--{-# INLINE cuTexRefGetAddressMode #-}-{# fun unsafe cuTexRefGetAddressMode-  { alloca-    `AddressMode' peekEnum*-  , useTexture `Texture'-  ,            `Int'                   } -> `Status' cToEnum #}----- |--- Get the filtering mode used by a texture reference.------ <http://docs.nvidia.com/cuda/cuda-driver-api/group__CUDA__TEXREF.html#group__CUDA__TEXREF_1g2439e069746f69b940f2f4dbc78cdf87>----{-# INLINEABLE getFilterMode #-}-getFilterMode :: Texture -> IO FilterMode-getFilterMode !tex = resultIfOk =<< cuTexRefGetFilterMode tex--{-# INLINE cuTexRefGetFilterMode #-}-{# fun unsafe cuTexRefGetFilterMode-  { alloca-    `FilterMode' peekEnum*-  , useTexture `Texture'              } -> `Status' cToEnum #}----- |--- Get the data format and number of channel components of the bound texture.------ <http://docs.nvidia.com/cuda/cuda-driver-api/group__CUDA__TEXREF.html#group__CUDA__TEXREF_1g90936eb6c7c4434a609e1160c278ae53>----{-# INLINEABLE getFormat #-}-getFormat :: Texture -> IO (Format, Int)-getFormat !tex = do-  (!status,!fmt,!dim) <- cuTexRefGetFormat tex-  resultIfOk (status,(fmt,dim))--{-# INLINE cuTexRefGetFormat #-}-{# fun unsafe cuTexRefGetFormat-  { alloca-    `Format'    peekEnum*-  , alloca-    `Int'       peekIntConv*-  , useTexture `Texture'                } -> `Status' cToEnum #}----- |--- Specify the addressing mode for the given dimension of a texture reference.------ <http://docs.nvidia.com/cuda/cuda-driver-api/group__CUDA__TEXREF.html#group__CUDA__TEXREF_1g85f4a13eeb94c8072f61091489349bcb>----{-# INLINEABLE setAddressMode #-}-setAddressMode :: Texture -> Int -> AddressMode -> IO ()-setAddressMode !tex !dim !mode = nothingIfOk =<< cuTexRefSetAddressMode tex dim mode--{-# INLINE cuTexRefSetAddressMode #-}-{# fun unsafe cuTexRefSetAddressMode-  { useTexture `Texture'-  ,            `Int'-  , cFromEnum  `AddressMode' } -> `Status' cToEnum #}----- |--- Specify the filtering mode to be used when reading memory through a texture--- reference.------ <http://docs.nvidia.com/cuda/cuda-driver-api/group__CUDA__TEXREF.html#group__CUDA__TEXREF_1g595d0af02c55576f8c835e4efd1f39c0>----{-# INLINEABLE setFilterMode #-}-setFilterMode :: Texture -> FilterMode -> IO ()-setFilterMode !tex !mode = nothingIfOk =<< cuTexRefSetFilterMode tex mode--{-# INLINE cuTexRefSetFilterMode #-}-{# fun unsafe cuTexRefSetFilterMode-  { useTexture `Texture'-  , cFromEnum  `FilterMode' } -> `Status' cToEnum #}----- |--- Specify additional characteristics for reading and indexing the texture--- reference.------ <http://docs.nvidia.com/cuda/cuda-driver-api/group__CUDA__TEXREF.html#group__CUDA__TEXREF_1g554ffd896487533c36810f2e45bb7a28>----{-# INLINEABLE setReadMode #-}-setReadMode :: Texture -> ReadMode -> IO ()-setReadMode !tex !mode = nothingIfOk =<< cuTexRefSetFlags tex mode--{-# INLINE cuTexRefSetFlags #-}-{# fun unsafe cuTexRefSetFlags-  { useTexture `Texture'-  , cFromEnum  `ReadMode' } -> `Status' cToEnum #}----- |--- Specify the format of the data and number of packed components per element to--- be read by the texture reference.------ <http://docs.nvidia.com/cuda/cuda-driver-api/group__CUDA__TEXREF.html#group__CUDA__TEXREF_1g05585ef8ea2fec728a03c6c8f87cf07a>----{-# INLINEABLE setFormat #-}-setFormat :: Texture -> Format -> Int -> IO ()-setFormat !tex !fmt !chn = nothingIfOk =<< cuTexRefSetFormat tex fmt chn--{-# INLINE cuTexRefSetFormat #-}-{# fun unsafe cuTexRefSetFormat-  { useTexture `Texture'-  , cFromEnum  `Format'-  ,            `Int'     } -> `Status' cToEnum #}-------------------------------------------------------------------------------------- Internal-----------------------------------------------------------------------------------{-# INLINE peekTex #-}-peekTex :: Ptr {# type CUtexref #} -> IO Texture-peekTex = liftM Texture . peek-
src/Foreign/CUDA/Runtime/Device.chs view
@@ -219,9 +219,15 @@ props !n = resultIfOk =<< cudaGetDeviceProperties n  {-# INLINE cudaGetDeviceProperties #-}+#if CUDA_VERSION < 12000 {# fun unsafe cudaGetDeviceProperties   { alloca- `DeviceProperties' peek*   ,         `Int'                    } -> `Status' cToEnum #}+#else+{# fun unsafe cudaGetDeviceProperties_v2 as cudaGetDeviceProperties+  { alloca- `DeviceProperties' peek*+  ,         `Int'                    } -> `Status' cToEnum #}+#endif   -- |
− src/Foreign/CUDA/Runtime/Texture.chs
@@ -1,203 +0,0 @@-{-# LANGUAGE BangPatterns             #-}-{-# LANGUAGE ForeignFunctionInterface #-}------------------------------------------------------------------------------------ |--- Module    : Foreign.CUDA.Runtime.Texture--- Copyright : [2009..2023] Trevor L. McDonell--- License   : BSD------ Texture references--------------------------------------------------------------------------------------module Foreign.CUDA.Runtime.Texture (--  -- * Texture Reference Management-  Texture(..), FormatKind(..), AddressMode(..), FilterMode(..), FormatDesc(..),-  bind, bind2D--) where---- Friends-import Foreign.CUDA.Ptr-import Foreign.CUDA.Runtime.Error-import Foreign.CUDA.Internal.C2HS---- System-import Data.Int-import Foreign-import Foreign.C--#include "cbits/stubs.h"-{# context lib="cudart" #}--#c-typedef struct textureReference      textureReference;-typedef struct cudaChannelFormatDesc cudaChannelFormatDesc;-#endc------------------------------------------------------------------------------------- Data Types------------------------------------------------------------------------------------- |A texture reference----{# pointer *textureReference as ^ -> Texture #}--data Texture = Texture-  {-    normalised :: !Bool,                -- ^ access texture using normalised coordinates [0.0,1.0)-    filtering  :: !FilterMode,-    addressing :: !(AddressMode, AddressMode, AddressMode),-    format     :: !FormatDesc-  }-  deriving (Eq, Show)---- |Texture channel format kind----{# enum cudaChannelFormatKind as FormatKind-  { }-  with prefix="cudaChannelFormatKind" deriving (Eq, Show) #}---- |Texture addressing mode----{# enum cudaTextureAddressMode as AddressMode-  { }-  with prefix="cudaAddressMode" deriving (Eq, Show) #}---- |Texture filtering mode----{# enum cudaTextureFilterMode as FilterMode-  { }-  with prefix="cudaFilterMode" deriving (Eq, Show) #}----- |A description of how memory read through the texture cache should be--- interpreted, including the kind of data and the number of bits of each--- component (x,y,z and w, respectively).----{# pointer *cudaChannelFormatDesc as ^ foreign -> FormatDesc nocode #}--data FormatDesc = FormatDesc-  {-    depth :: !(Int,Int,Int,Int),-    kind  :: !FormatKind-  }-  deriving (Eq, Show)--instance Storable FormatDesc where-  sizeOf    _ = {# sizeof cudaChannelFormatDesc #}-  alignment _ = alignment (undefined :: Ptr ())--  peek p = do-    dx <- cIntConv `fmap` {# get cudaChannelFormatDesc.x #} p-    dy <- cIntConv `fmap` {# get cudaChannelFormatDesc.y #} p-    dz <- cIntConv `fmap` {# get cudaChannelFormatDesc.z #} p-    dw <- cIntConv `fmap` {# get cudaChannelFormatDesc.w #} p-    df <- cToEnum  `fmap` {# get cudaChannelFormatDesc.f #} p-    return $ FormatDesc (dx,dy,dz,dw) df--  poke p (FormatDesc (x,y,z,w) k) = do-    {# set cudaChannelFormatDesc.x #} p (cIntConv x)-    {# set cudaChannelFormatDesc.y #} p (cIntConv y)-    {# set cudaChannelFormatDesc.z #} p (cIntConv z)-    {# set cudaChannelFormatDesc.w #} p (cIntConv w)-    {# set cudaChannelFormatDesc.f #} p (cFromEnum k)---instance Storable Texture where-  sizeOf    _ = {# sizeof textureReference #}-  alignment _ = alignment (undefined :: Ptr ())--  peek p = do-    norm    <- cToBool `fmap` {# get textureReference.normalized #} p-    fmt     <- cToEnum `fmap` {# get textureReference.filterMode #} p-    dsc     <- peek . castPtr          =<< {# get textureReference.channelDesc #} p-    [x,y,z] <- peekArrayWith cToEnum 3 =<< {# get textureReference.addressMode #} p-    return $ Texture norm fmt (x,y,z) dsc--  poke p (Texture norm fmt (x,y,z) dsc) = do-    {# set textureReference.normalized #} p (cFromBool norm)-    {# set textureReference.filterMode #} p (cFromEnum fmt)-    withArray (map cFromEnum [x,y,z]) ({# set textureReference.addressMode #} p)--    -- c2hs is returning the wrong type for structs-within-structs-    dscptr <- {# get textureReference.channelDesc #} p-    poke (castPtr dscptr) dsc-------------------------------------------------------------------------------------- Texture References------------------------------------------------------------------------------------- |Bind the memory area associated with the device pointer to a texture--- reference given by the named symbol. Any previously bound references are--- unbound.----{-# INLINEABLE bind #-}-bind :: String -> Texture -> DevicePtr a -> Int64 -> IO ()-bind !name !tex !dptr !bytes = do-  ref <- getTex name-  poke ref tex-  nothingIfOk =<< cudaBindTexture ref dptr (format tex) bytes--{-# INLINE cudaBindTexture #-}-{# fun unsafe cudaBindTexture-  { alloca- `Int'-  , id      `TextureReference'-  , dptr    `DevicePtr a'-  , with_*  `FormatDesc'-  ,         `Int64'            } -> `Status' cToEnum #}-  where dptr = useDevicePtr . castDevPtr---- |Bind the two-dimensional memory area to the texture reference associated--- with the given symbol. The size of the area is constrained by (width,height)--- in texel units, and the row pitch in bytes. Any previously bound references--- are unbound.----{-# INLINEABLE bind2D #-}-bind2D :: String -> Texture -> DevicePtr a -> (Int,Int) -> Int64 -> IO ()-bind2D !name !tex !dptr (!width,!height) !bytes = do-  ref <- getTex name-  poke ref tex-  nothingIfOk =<< cudaBindTexture2D ref dptr (format tex) width height bytes--{-# INLINE cudaBindTexture2D #-}-{# fun unsafe cudaBindTexture2D-  { alloca- `Int'-  , id      `TextureReference'-  , dptr    `DevicePtr a'-  , with_*  `FormatDesc'-  ,         `Int'-  ,         `Int'-  ,         `Int64'             } -> `Status' cToEnum #}-  where dptr = useDevicePtr . castDevPtr----- |Returns the texture reference associated with the given symbol----{-# INLINEABLE getTex #-}-getTex :: String -> IO TextureReference-getTex !name = resultIfOk =<< cudaGetTextureReference name--{-# INLINE cudaGetTextureReference #-}-{# fun unsafe cudaGetTextureReference-  { alloca-       `Ptr Texture' peek*-  , withCString_* `String'            } -> `Status' cToEnum #}-------------------------------------------------------------------------------------- Internal-----------------------------------------------------------------------------------{-# INLINE with_ #-}-with_ :: Storable a => a -> (Ptr a -> IO b) -> IO b-with_ = with----- CUDA 5.0 changed the types of some attributes from char* to void*----{-# INLINE withCString_ #-}-withCString_ :: String -> (Ptr a -> IO b) -> IO b-withCString_ !str !fn = withCString str (fn . castPtr)-