dataframe-core 2.4.0.0 → 2.5.0.0
raw patch · 62 files changed
+11479/−7216 lines, 62 filesPVP ok
version bump matches the API change (PVP)
API changes (from Hackage documentation)
- DataFrame.Internal.AggKernel: RCount :: Reduction
- DataFrame.Internal.AggKernel: RMax :: Reduction
- DataFrame.Internal.AggKernel: RMean :: Reduction
- DataFrame.Internal.AggKernel: RMin :: Reduction
- DataFrame.Internal.AggKernel: RStd :: Reduction
- DataFrame.Internal.AggKernel: RSum :: Reduction
- DataFrame.Internal.AggKernel: RTop2Sum :: Reduction
- DataFrame.Internal.AggKernel: RVar :: Reduction
- DataFrame.Internal.AggKernel: data Reduction
- DataFrame.Internal.AggKernel: instance GHC.Classes.Eq DataFrame.Internal.AggKernel.Reduction
- DataFrame.Internal.AggKernel: instance GHC.Show.Show DataFrame.Internal.AggKernel.Reduction
- DataFrame.Internal.AggKernel: scatterColumnToDouble :: Column -> Maybe (Vector Double)
- DataFrame.Internal.AggKernel: scatterReduce :: Reduction -> Vector Int -> Int -> Column -> Maybe Column
- DataFrame.Internal.AggKernelDirect: directReduce :: Reduction -> Vector Int -> Int -> Column -> Maybe Column
- DataFrame.Internal.AggKernelDirect: directThreshold :: Int
- DataFrame.Internal.AggKernelPar: momentScatterPar :: Vector Int -> Vector Int -> Int -> Column -> Column -> Maybe Moments
- DataFrame.Internal.AggKernelPar: scatterReducePar :: Reduction -> Vector Int -> Vector Int -> Int -> Column -> Maybe Column
- DataFrame.Internal.AggPlan: MomentPlan :: Text -> Text -> Text -> Text -> Text -> Text -> Text -> Text -> MomentPlan
- DataFrame.Internal.AggPlan: Moments :: Column -> Column -> Column -> Column -> Column -> Column -> Moments
- DataFrame.Internal.AggPlan: PlanMaxMinusMin :: Text -> Text -> AggPlan
- DataFrame.Internal.AggPlan: PlanMedian :: Text -> AggPlan
- DataFrame.Internal.AggPlan: PlanScatter :: Reduction -> Text -> AggPlan
- DataFrame.Internal.AggPlan: [mN] :: Moments -> Column
- DataFrame.Internal.AggPlan: [mSx] :: Moments -> Column
- DataFrame.Internal.AggPlan: [mSxx] :: Moments -> Column
- DataFrame.Internal.AggPlan: [mSxy] :: Moments -> Column
- DataFrame.Internal.AggPlan: [mSy] :: Moments -> Column
- DataFrame.Internal.AggPlan: [mSyy] :: Moments -> Column
- DataFrame.Internal.AggPlan: [mpColX] :: MomentPlan -> Text
- DataFrame.Internal.AggPlan: [mpColY] :: MomentPlan -> Text
- DataFrame.Internal.AggPlan: [mpNName] :: MomentPlan -> Text
- DataFrame.Internal.AggPlan: [mpSxName] :: MomentPlan -> Text
- DataFrame.Internal.AggPlan: [mpSxxName] :: MomentPlan -> Text
- DataFrame.Internal.AggPlan: [mpSxyName] :: MomentPlan -> Text
- DataFrame.Internal.AggPlan: [mpSyName] :: MomentPlan -> Text
- DataFrame.Internal.AggPlan: [mpSyyName] :: MomentPlan -> Text
- DataFrame.Internal.AggPlan: data AggPlan
- DataFrame.Internal.AggPlan: data MomentPlan
- DataFrame.Internal.AggPlan: data Moments
- DataFrame.Internal.AggPlan: instance GHC.Classes.Eq DataFrame.Internal.AggPlan.Role
- DataFrame.Internal.AggPlan: instance GHC.Classes.Eq DataFrame.Internal.AggPlan.Term
- DataFrame.Internal.AggPlan: instance GHC.Classes.Ord DataFrame.Internal.AggPlan.Role
- DataFrame.Internal.AggPlan: instance GHC.Classes.Ord DataFrame.Internal.AggPlan.Term
- DataFrame.Internal.AggPlan: instance GHC.Show.Show DataFrame.Internal.AggPlan.Role
- DataFrame.Internal.AggPlan: instance GHC.Show.Show DataFrame.Internal.AggPlan.Term
- DataFrame.Internal.AggPlan: momentScatter :: Vector Int -> Int -> Column -> Column -> Maybe Moments
- DataFrame.Internal.AggPlan: planAgg :: GroupedDataFrame -> UExpr -> Maybe AggPlan
- DataFrame.Internal.AggPlan: planMoments :: GroupedDataFrame -> [(Text, UExpr)] -> Maybe MomentPlan
- DataFrame.Internal.Column: [BoxedColumn] :: forall a. Columnable a => Maybe Bitmap -> Vector a -> Column
- DataFrame.Internal.Column: [MBoxedColumn] :: forall a. Columnable a => IOVector a -> MutableColumn
- DataFrame.Internal.Column: [MUnboxedColumn] :: forall a. (Columnable a, Unbox a) => IOVector a -> MutableColumn
- DataFrame.Internal.Column: [MergedColumn] :: !Column -> !Column -> Column
- DataFrame.Internal.Column: [PackedText] :: Maybe Bitmap -> {-# UNPACK #-} !PackedTextData -> Column
- DataFrame.Internal.Column: [TColumn] :: forall a. Columnable a => Column -> TypedColumn a
- DataFrame.Internal.Column: [UnboxedColumn] :: forall a. (Columnable a, Unbox a) => Maybe Bitmap -> Vector a -> Column
- DataFrame.Internal.Column: allMissing :: Column -> Bool
- DataFrame.Internal.Column: allValidBitmap :: Int -> Bitmap
- DataFrame.Internal.Column: atIndicesStable :: Vector Int -> Column -> Column
- DataFrame.Internal.Column: bitmapConcat :: Int -> Bitmap -> Int -> Bitmap -> Bitmap
- DataFrame.Internal.Column: bitmapSlice :: Int -> Int -> Bitmap -> Bitmap
- DataFrame.Internal.Column: bitmapTestBit :: Bitmap -> Int -> Bool
- DataFrame.Internal.Column: buildBitmapFromNulls :: Int -> [Int] -> Bitmap
- DataFrame.Internal.Column: buildBitmapFromValid :: Vector Word8 -> Bitmap
- DataFrame.Internal.Column: checkMergedNoBothNull :: Column -> Column -> ()
- DataFrame.Internal.Column: class ColumnifyRep (r :: Rep) a
- DataFrame.Internal.Column: columnBitmap :: Column -> Maybe Bitmap
- DataFrame.Internal.Column: columnElemIsNull :: Column -> Int -> Bool
- DataFrame.Internal.Column: columnLength :: Column -> Int
- DataFrame.Internal.Column: columnTypeString :: Column -> String
- DataFrame.Internal.Column: columnVersionString :: Column -> String
- DataFrame.Internal.Column: concatColumns :: Column -> Column -> Either DataFrameException Column
- DataFrame.Internal.Column: concatColumnsEither :: Column -> Column -> Column
- DataFrame.Internal.Column: concatManyColumns :: [Column] -> Column
- DataFrame.Internal.Column: copyIntoMutableColumn :: MutableColumn -> Int -> Column -> IO ()
- DataFrame.Internal.Column: data Column
- DataFrame.Internal.Column: data MutableColumn
- DataFrame.Internal.Column: data TypedColumn a
- DataFrame.Internal.Column: ensureOptional :: Column -> Column
- DataFrame.Internal.Column: eqBoxedCols :: Eq a => Maybe Bitmap -> Vector a -> Maybe Bitmap -> Vector a -> Bool
- DataFrame.Internal.Column: eqPackedCols :: Maybe Bitmap -> PackedTextData -> Maybe Bitmap -> PackedTextData -> Bool
- DataFrame.Internal.Column: expandColumn :: Int -> Column -> Column
- DataFrame.Internal.Column: finalizeParseResult :: Unbox a => STVector s a -> STVector s Word8 -> Bool -> ST s (Maybe (Maybe Bitmap, Vector a))
- DataFrame.Internal.Column: findIndices :: Columnable a => (a -> Bool) -> Column -> Either DataFrameException (Vector Int)
- DataFrame.Internal.Column: foldLinearGroups :: (Columnable b, Columnable acc) => (acc -> b -> acc) -> acc -> Column -> Vector Int -> Int -> Either DataFrameException Column
- DataFrame.Internal.Column: foldl1Column :: Columnable a => (a -> a -> a) -> Column -> Either DataFrameException a
- DataFrame.Internal.Column: foldl1DirectGroups :: Columnable a => (a -> a -> a) -> Column -> Vector Int -> Vector Int -> Either DataFrameException Column
- DataFrame.Internal.Column: foldlColumn :: (Columnable a, Columnable b) => (b -> a -> b) -> b -> Column -> Either DataFrameException b
- DataFrame.Internal.Column: forceColumn :: Column -> ()
- DataFrame.Internal.Column: freezeColumnEither :: [(Int, Text)] -> MutableColumn -> IO Column
- DataFrame.Internal.Column: freezeMutableColumn :: MutableColumn -> IO Column
- DataFrame.Internal.Column: fromList :: (Columnable a, ColumnifyRep (KindOf a) a) => [a] -> Column
- DataFrame.Internal.Column: fromMaybeVec :: Columnable a => Vector (Maybe a) -> Column
- DataFrame.Internal.Column: fromMaybeVecUnboxed :: (Columnable a, Unbox a) => Vector (Maybe a) -> Column
- DataFrame.Internal.Column: fromUnboxedVector :: (Columnable a, Unbox a) => Vector a -> Column
- DataFrame.Internal.Column: fromVector :: (Columnable a, ColumnifyRep (KindOf a) a) => Vector a -> Column
- DataFrame.Internal.Column: gatherWithSentinel :: Vector Int -> Column -> Column
- DataFrame.Internal.Column: getIndices :: Vector Int -> Vector a -> Vector a
- DataFrame.Internal.Column: getIndicesUnboxed :: Unbox a => Vector Int -> Vector a -> Vector a
- DataFrame.Internal.Column: hasElemType :: Columnable a => Column -> Bool
- DataFrame.Internal.Column: hasMissing :: Column -> Bool
- DataFrame.Internal.Column: headColumn :: Columnable a => Column -> Either DataFrameException a
- DataFrame.Internal.Column: ifoldrColumn :: (Columnable a, Columnable b) => (Int -> a -> b -> b) -> b -> Column -> Either DataFrameException b
- DataFrame.Internal.Column: imapColumn :: (Columnable b, Columnable c) => (Int -> b -> c) -> Column -> Either DataFrameException Column
- DataFrame.Internal.Column: instance (DataFrame.Internal.Column.Columnable a, Data.Vector.Unboxed.Base.Unbox a) => DataFrame.Internal.Column.ColumnifyRep 'DataFrame.Internal.Types.RUnboxed a
- DataFrame.Internal.Column: instance DataFrame.Internal.Column.Columnable a => DataFrame.Internal.Column.ColumnifyRep 'DataFrame.Internal.Types.RBoxed a
- DataFrame.Internal.Column: instance DataFrame.Internal.Column.Columnable a => DataFrame.Internal.Column.ColumnifyRep 'DataFrame.Internal.Types.RNullableBoxed (GHC.Maybe.Maybe a)
- DataFrame.Internal.Column: instance GHC.Classes.Eq DataFrame.Internal.Column.Column
- DataFrame.Internal.Column: instance GHC.Classes.Eq a => GHC.Classes.Eq (DataFrame.Internal.Column.TypedColumn a)
- DataFrame.Internal.Column: instance GHC.Show.Show DataFrame.Internal.Column.Column
- DataFrame.Internal.Column: instance GHC.Show.Show a => GHC.Show.Show (DataFrame.Internal.Column.TypedColumn a)
- DataFrame.Internal.Column: isMergedColumn :: Column -> Bool
- DataFrame.Internal.Column: isNumeric :: Column -> Bool
- DataFrame.Internal.Column: isPackedText :: Column -> Bool
- DataFrame.Internal.Column: leftExpandColumn :: Int -> Column -> Column
- DataFrame.Internal.Column: mapColumn :: (Columnable b, Columnable c) => (b -> c) -> Column -> Either DataFrameException Column
- DataFrame.Internal.Column: materializeMerged :: Column -> Column
- DataFrame.Internal.Column: materializePacked :: Column -> Column
- DataFrame.Internal.Column: mergeBitmaps :: Bitmap -> Bitmap -> Bitmap
- DataFrame.Internal.Column: mergeColumns :: Column -> Column -> Column
- DataFrame.Internal.Column: mergeEager :: Column -> Column -> Column
- DataFrame.Internal.Column: mergedHead :: Column -> Column
- DataFrame.Internal.Column: mkRandom :: (RandomGen g, Columnable a, ColumnifyRep (KindOf a) a, UniformRange a) => g -> Int -> a -> a -> Column
- DataFrame.Internal.Column: newMutableColumn :: Int -> Column -> IO MutableColumn
- DataFrame.Internal.Column: numElements :: Column -> Int
- DataFrame.Internal.Column: sliceColumn :: Int -> Int -> Column -> Column
- DataFrame.Internal.Column: takeColumn :: Int -> Column -> Column
- DataFrame.Internal.Column: takeLastColumn :: Int -> Column -> Column
- DataFrame.Internal.Column: throwTypeMismatch :: (Typeable a, Typeable b) => Either DataFrameException Column
- DataFrame.Internal.Column: toColumnRep :: ColumnifyRep r a => Vector a -> Column
- DataFrame.Internal.Column: toDoubleVector :: Column -> Either DataFrameException (Vector Double)
- DataFrame.Internal.Column: toFloatVector :: Column -> Either DataFrameException (Vector Float)
- DataFrame.Internal.Column: toIntVector :: Column -> Either DataFrameException (Vector Int)
- DataFrame.Internal.Column: toList :: Columnable a => Column -> [a]
- DataFrame.Internal.Column: toUnboxedVector :: (Columnable a, Unbox a) => Column -> Either DataFrameException (Vector a)
- DataFrame.Internal.Column: toVector :: forall a v. (Vector v a, Columnable a) => Column -> Either DataFrameException (v a)
- DataFrame.Internal.Column: type Bitmap = Vector Word8
- DataFrame.Internal.Column: type Columnable a = (Columnable' a, ColumnifyRep KindOf a a, UnboxIf a, IntegralIf a, FloatingIf a, SBoolI Unboxable a, SBoolI Numeric a, SBoolI IntegralTypes a, SBoolI FloatingTypes a)
- DataFrame.Internal.Column: unwrapTypedColumn :: TypedColumn a -> Column
- DataFrame.Internal.Column: vectorFromTypedColumn :: TypedColumn a -> Vector a
- DataFrame.Internal.Column: zipColumns :: Column -> Column -> Column
- DataFrame.Internal.Column: zipWithColumns :: (Columnable a, Columnable b, Columnable c) => (a -> b -> c) -> Column -> Column -> Either DataFrameException Column
- DataFrame.Internal.Column: zipWithColumnsGeneral :: (Columnable a, Columnable b, Columnable c) => (a -> b -> c) -> Column -> Column -> Either DataFrameException Column
- DataFrame.Internal.ColumnBuilder: TextChunk :: !Array -> !Int -> !Vector Int -> !Maybe Bitmap -> TextChunk
- DataFrame.Internal.ColumnBuilder: [tcBitmap] :: TextChunk -> !Maybe Bitmap
- DataFrame.Internal.ColumnBuilder: [tcBytes] :: TextChunk -> !Array
- DataFrame.Internal.ColumnBuilder: [tcOffsets] :: TextChunk -> !Vector Int
- DataFrame.Internal.ColumnBuilder: [tcUsed] :: TextChunk -> !Int
- DataFrame.Internal.ColumnBuilder: appendDouble :: DoubleBuilder s -> Double -> ST s ()
- DataFrame.Internal.ColumnBuilder: appendInt :: IntBuilder s -> Int -> ST s ()
- DataFrame.Internal.ColumnBuilder: appendNull :: ColumnBuilder b => b s -> ST s ()
- DataFrame.Internal.ColumnBuilder: appendNum :: Unbox a => NumBuilder a s -> a -> ST s ()
- DataFrame.Internal.ColumnBuilder: appendText :: TextBuilder s -> Text -> ST s ()
- DataFrame.Internal.ColumnBuilder: appendTextSlice :: TextBuilder s -> Array -> Int -> Int -> ST s ()
- DataFrame.Internal.ColumnBuilder: appendTextSliceFromPtr :: TextBuilder s -> Ptr Word8 -> Int -> ST s ()
- DataFrame.Internal.ColumnBuilder: builderLength :: ColumnBuilder b => b s -> ST s Int
- DataFrame.Internal.ColumnBuilder: class ColumnBuilder (b :: Type -> Type)
- DataFrame.Internal.ColumnBuilder: data NumBuilder a s
- DataFrame.Internal.ColumnBuilder: data TextBuilder s
- DataFrame.Internal.ColumnBuilder: data TextChunk
- DataFrame.Internal.ColumnBuilder: freezeBuilder :: ColumnBuilder b => b s -> ST s Column
- DataFrame.Internal.ColumnBuilder: freezeTextChunk :: TextBuilder s -> ST s TextChunk
- DataFrame.Internal.ColumnBuilder: instance (DataFrame.Internal.Column.Columnable a, Data.Vector.Unboxed.Base.Unbox a) => DataFrame.Internal.ColumnBuilder.ColumnBuilder (DataFrame.Internal.ColumnBuilder.NumBuilder a)
- DataFrame.Internal.ColumnBuilder: instance DataFrame.Internal.ColumnBuilder.ColumnBuilder DataFrame.Internal.ColumnBuilder.TextBuilder
- DataFrame.Internal.ColumnBuilder: mergeColumns :: [Column] -> Column
- DataFrame.Internal.ColumnBuilder: mergeTextChunks :: [TextChunk] -> Column
- DataFrame.Internal.ColumnBuilder: newDoubleBuilder :: Int -> ST s (DoubleBuilder s)
- DataFrame.Internal.ColumnBuilder: newIntBuilder :: Int -> ST s (IntBuilder s)
- DataFrame.Internal.ColumnBuilder: newNumBuilder :: Unbox a => a -> Int -> ST s (NumBuilder a s)
- DataFrame.Internal.ColumnBuilder: newTextBuilder :: Int -> Int -> ST s (TextBuilder s)
- DataFrame.Internal.ColumnBuilder: type DoubleBuilder = NumBuilder Double
- DataFrame.Internal.ColumnBuilder: type IntBuilder = NumBuilder Int
- DataFrame.Internal.ColumnMerge: TextChunk :: !Array -> !Int -> !Vector Int -> !Maybe Bitmap -> TextChunk
- DataFrame.Internal.ColumnMerge: [tcBitmap] :: TextChunk -> !Maybe Bitmap
- DataFrame.Internal.ColumnMerge: [tcBytes] :: TextChunk -> !Array
- DataFrame.Internal.ColumnMerge: [tcOffsets] :: TextChunk -> !Vector Int
- DataFrame.Internal.ColumnMerge: [tcUsed] :: TextChunk -> !Int
- DataFrame.Internal.ColumnMerge: data TextChunk
- DataFrame.Internal.ColumnMerge: mergeColumns :: [Column] -> Column
- DataFrame.Internal.ColumnMerge: mergeTextChunks :: [TextChunk] -> Column
- DataFrame.Internal.ColumnMerge: packValidity :: Int -> MVector s Word8 -> ST s Bitmap
- DataFrame.Internal.ColumnMerge: packedFromTextChunk :: TextChunk -> Column
- DataFrame.Internal.ColumnMerge: spliceBitmaps :: [(Maybe Bitmap, Int)] -> Maybe Bitmap
- DataFrame.Internal.ColumnMerge: tcRows :: TextChunk -> Int
- DataFrame.Internal.DataFrame: Grouped :: DataFrame -> [Text] -> Vector Int -> Vector Int -> Vector Int -> GroupedDataFrame
- DataFrame.Internal.DictEncode: dictCompactColumn :: Column -> Column
- DataFrame.Internal.DictEncode: dictEncodeColumn :: Column -> Maybe (Vector Int, Int)
- DataFrame.Internal.DictEncode: dictEncodeColumnUpTo :: Int -> Column -> Maybe (Vector Int, Int)
- DataFrame.Internal.DictEncode: dictMaxCardinality :: Int
- DataFrame.Internal.Expression: instance (Data.String.IsString a, DataFrame.Internal.Column.Columnable a) => Data.String.IsString (DataFrame.Internal.Expression.Expr a)
- DataFrame.Internal.Expression: instance (GHC.Float.Floating a, DataFrame.Internal.Column.Columnable a) => GHC.Float.Floating (DataFrame.Internal.Expression.Expr a)
- DataFrame.Internal.Expression: instance (GHC.Num.Num a, DataFrame.Internal.Column.Columnable a) => GHC.Num.Num (DataFrame.Internal.Expression.Expr a)
- DataFrame.Internal.Expression: instance (GHC.Real.Fractional a, DataFrame.Internal.Column.Columnable a) => GHC.Real.Fractional (DataFrame.Internal.Expression.Expr a)
- DataFrame.Internal.GroupingDirect: DirectGrouping :: !Vector Int -> !Vector Int -> !Vector Int -> !Int -> DirectGrouping
- DataFrame.Internal.GroupingDirect: [dgNGroups] :: DirectGrouping -> !Int
- DataFrame.Internal.GroupingDirect: [dgOffsets] :: DirectGrouping -> !Vector Int
- DataFrame.Internal.GroupingDirect: [dgRowToGroup] :: DirectGrouping -> !Vector Int
- DataFrame.Internal.GroupingDirect: [dgValueIndices] :: DirectGrouping -> !Vector Int
- DataFrame.Internal.GroupingDirect: data DirectGrouping
- DataFrame.Internal.GroupingDirect: directGroupThreshold :: Int
- DataFrame.Internal.GroupingDirect: tryDirectGroupColumn :: Column -> Maybe DirectGrouping
- DataFrame.Internal.GroupingPar: numPartitionsFor :: Int -> Int
- DataFrame.Internal.GroupingPar: parThreshold :: Int
- DataFrame.Internal.GroupingPar: parallelAssignGroups :: Int -> Vector Int -> (Int -> Int -> Bool) -> IO (Vector Int, Vector Int, Vector Int)
- DataFrame.Internal.GroupingPar: shouldParallelize :: Int -> Bool
- DataFrame.Internal.Hash: fnvOffset :: Int
- DataFrame.Internal.Hash: mixBool :: Int -> Bool -> Int
- DataFrame.Internal.Hash: mixBytes :: Int -> Array -> Int -> Int -> Int
- DataFrame.Internal.Hash: mixChar :: Int -> Char -> Int
- DataFrame.Internal.Hash: mixDouble :: Int -> Double -> Int
- DataFrame.Internal.Hash: mixInt :: Int -> Int -> Int
- DataFrame.Internal.Hash: mixShow :: Show a => Int -> a -> Int
- DataFrame.Internal.Hash: mixText :: Int -> Text -> Int
- DataFrame.Internal.Hash: nullSalt :: Int
- DataFrame.Internal.HashTable: HashTable :: !MVector s Int -> !MVector s Int -> !MVector s Int -> !Int -> HashTable s
- DataFrame.Internal.HashTable: [htGroup] :: HashTable s -> !MVector s Int
- DataFrame.Internal.HashTable: [htHash] :: HashTable s -> !MVector s Int
- DataFrame.Internal.HashTable: [htMask] :: HashTable s -> !Int
- DataFrame.Internal.HashTable: [htRep] :: HashTable s -> !MVector s Int
- DataFrame.Internal.HashTable: data HashTable s
- DataFrame.Internal.HashTable: htInsert :: PrimMonad m => HashTable (PrimState m) -> (Int -> Int -> Bool) -> Int -> Int -> Int -> m (Int, Bool)
- DataFrame.Internal.HashTable: newHashTable :: PrimMonad m => Int -> m (HashTable (PrimState m))
- DataFrame.Internal.HashTable: nextPow2Above :: Int -> Int
- DataFrame.Internal.Nullable: applyNull1 :: NullLift1Op a r c => (BaseType a -> r) -> a -> c
- DataFrame.Internal.Nullable: applyNull2 :: NullLift2Op a b r c => (BaseType a -> BaseType b -> r) -> a -> b -> c
- DataFrame.Internal.Nullable: class Columnable PromoteDiv a b => DivWidenOp a b
- DataFrame.Internal.Nullable: class (Columnable a, Columnable r, Columnable c) => NullLift1Op a r c
- DataFrame.Internal.Nullable: class (Columnable a, Columnable b, Columnable r, Columnable c) => NullLift2Op a b r c
- DataFrame.Internal.Nullable: class (Columnable a, Columnable b, Columnable c) => NullableArithOp a b c | a b -> c
- DataFrame.Internal.Nullable: class (Columnable a, Columnable b, Columnable e) => NullableCmpOp a b e
- DataFrame.Internal.Nullable: class Columnable Promote a b => NumericWidenOp a b
- DataFrame.Internal.Nullable: divArithOp :: DivWidenOp a b => (PromoteDiv a b -> PromoteDiv a b -> PromoteDiv a b) -> a -> b -> PromoteDiv a b
- DataFrame.Internal.Nullable: divWiden1 :: DivWidenOp a b => a -> PromoteDiv a b
- DataFrame.Internal.Nullable: divWiden2 :: DivWidenOp a b => b -> PromoteDiv a b
- DataFrame.Internal.Nullable: instance (DataFrame.Internal.Column.Columnable a, DataFrame.Internal.Column.Columnable (GHC.Maybe.Maybe a)) => DataFrame.Internal.Nullable.NullableArithOp (GHC.Maybe.Maybe a) (GHC.Maybe.Maybe a) (GHC.Maybe.Maybe a)
- DataFrame.Internal.Nullable: instance (DataFrame.Internal.Column.Columnable a, DataFrame.Internal.Column.Columnable (GHC.Maybe.Maybe a)) => DataFrame.Internal.Nullable.NullableArithOp (GHC.Maybe.Maybe a) a (GHC.Maybe.Maybe a)
- DataFrame.Internal.Nullable: instance (DataFrame.Internal.Column.Columnable a, DataFrame.Internal.Column.Columnable (GHC.Maybe.Maybe a), DataFrame.Internal.Column.Columnable (GHC.Maybe.Maybe GHC.Types.Bool)) => DataFrame.Internal.Nullable.NullableCmpOp (GHC.Maybe.Maybe a) (GHC.Maybe.Maybe a) (GHC.Maybe.Maybe GHC.Types.Bool)
- DataFrame.Internal.Nullable: instance (DataFrame.Internal.Column.Columnable a, DataFrame.Internal.Column.Columnable (GHC.Maybe.Maybe a), DataFrame.Internal.Column.Columnable (GHC.Maybe.Maybe GHC.Types.Bool)) => DataFrame.Internal.Nullable.NullableCmpOp (GHC.Maybe.Maybe a) a (GHC.Maybe.Maybe GHC.Types.Bool)
- DataFrame.Internal.Nullable: instance (DataFrame.Internal.Column.Columnable a, DataFrame.Internal.Column.Columnable (GHC.Maybe.Maybe a), DataFrame.Internal.Column.Columnable (GHC.Maybe.Maybe GHC.Types.Bool), a GHC.Types.~ DataFrame.Internal.Nullable.BaseType a) => DataFrame.Internal.Nullable.NullableCmpOp a (GHC.Maybe.Maybe a) (GHC.Maybe.Maybe GHC.Types.Bool)
- DataFrame.Internal.Nullable: instance (DataFrame.Internal.Column.Columnable a, DataFrame.Internal.Column.Columnable (GHC.Maybe.Maybe a), a GHC.Types.~ DataFrame.Internal.Nullable.BaseType a) => DataFrame.Internal.Nullable.NullableArithOp a (GHC.Maybe.Maybe a) (GHC.Maybe.Maybe a)
- DataFrame.Internal.Nullable: instance (DataFrame.Internal.Column.Columnable a, DataFrame.Internal.Column.Columnable GHC.Types.Bool, a GHC.Types.~ DataFrame.Internal.Nullable.BaseType a) => DataFrame.Internal.Nullable.NullableCmpOp a a GHC.Types.Bool
- DataFrame.Internal.Nullable: instance (DataFrame.Internal.Column.Columnable a, DataFrame.Internal.Column.Columnable b, DataFrame.Internal.Column.Columnable r, DataFrame.Internal.Column.Columnable (GHC.Maybe.Maybe r)) => DataFrame.Internal.Nullable.NullLift2Op (GHC.Maybe.Maybe a) (GHC.Maybe.Maybe b) r (GHC.Maybe.Maybe r)
- DataFrame.Internal.Nullable: instance (DataFrame.Internal.Column.Columnable a, DataFrame.Internal.Column.Columnable b, DataFrame.Internal.Column.Columnable r, DataFrame.Internal.Column.Columnable (GHC.Maybe.Maybe r), a GHC.Types.~ DataFrame.Internal.Nullable.BaseType a) => DataFrame.Internal.Nullable.NullLift2Op a (GHC.Maybe.Maybe b) r (GHC.Maybe.Maybe r)
- DataFrame.Internal.Nullable: instance (DataFrame.Internal.Column.Columnable a, DataFrame.Internal.Column.Columnable b, DataFrame.Internal.Column.Columnable r, DataFrame.Internal.Column.Columnable (GHC.Maybe.Maybe r), b GHC.Types.~ DataFrame.Internal.Nullable.BaseType b) => DataFrame.Internal.Nullable.NullLift2Op (GHC.Maybe.Maybe a) b r (GHC.Maybe.Maybe r)
- DataFrame.Internal.Nullable: instance (DataFrame.Internal.Column.Columnable a, DataFrame.Internal.Column.Columnable b, DataFrame.Internal.Column.Columnable r, a GHC.Types.~ DataFrame.Internal.Nullable.BaseType a, b GHC.Types.~ DataFrame.Internal.Nullable.BaseType b) => DataFrame.Internal.Nullable.NullLift2Op a b r r
- DataFrame.Internal.Nullable: instance (DataFrame.Internal.Column.Columnable a, DataFrame.Internal.Column.Columnable r, DataFrame.Internal.Column.Columnable (GHC.Maybe.Maybe r)) => DataFrame.Internal.Nullable.NullLift1Op (GHC.Maybe.Maybe a) r (GHC.Maybe.Maybe r)
- DataFrame.Internal.Nullable: instance (DataFrame.Internal.Column.Columnable a, DataFrame.Internal.Column.Columnable r, a GHC.Types.~ DataFrame.Internal.Nullable.BaseType a) => DataFrame.Internal.Nullable.NullLift1Op a r r
- DataFrame.Internal.Nullable: instance (DataFrame.Internal.Column.Columnable a, a GHC.Types.~ DataFrame.Internal.Nullable.BaseType a) => DataFrame.Internal.Nullable.NullableArithOp a a a
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Column.Columnable a => DataFrame.Internal.Nullable.NumericWidenOp a a
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.DivWidenOp GHC.Int.Int32 GHC.Int.Int32
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.DivWidenOp GHC.Int.Int32 GHC.Int.Int64
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.DivWidenOp GHC.Int.Int32 GHC.Types.Double
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.DivWidenOp GHC.Int.Int32 GHC.Types.Float
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.DivWidenOp GHC.Int.Int32 GHC.Types.Int
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.DivWidenOp GHC.Int.Int64 GHC.Int.Int32
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.DivWidenOp GHC.Int.Int64 GHC.Int.Int64
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.DivWidenOp GHC.Int.Int64 GHC.Types.Double
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.DivWidenOp GHC.Int.Int64 GHC.Types.Float
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.DivWidenOp GHC.Int.Int64 GHC.Types.Int
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.DivWidenOp GHC.Types.Double GHC.Int.Int32
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.DivWidenOp GHC.Types.Double GHC.Int.Int64
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.DivWidenOp GHC.Types.Double GHC.Types.Double
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.DivWidenOp GHC.Types.Double GHC.Types.Float
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.DivWidenOp GHC.Types.Double GHC.Types.Int
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.DivWidenOp GHC.Types.Float GHC.Int.Int32
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.DivWidenOp GHC.Types.Float GHC.Int.Int64
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.DivWidenOp GHC.Types.Float GHC.Types.Double
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.DivWidenOp GHC.Types.Float GHC.Types.Float
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.DivWidenOp GHC.Types.Float GHC.Types.Int
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.DivWidenOp GHC.Types.Int GHC.Int.Int32
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.DivWidenOp GHC.Types.Int GHC.Int.Int64
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.DivWidenOp GHC.Types.Int GHC.Types.Double
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.DivWidenOp GHC.Types.Int GHC.Types.Float
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.DivWidenOp GHC.Types.Int GHC.Types.Int
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.NumericWidenOp GHC.Int.Int32 GHC.Types.Double
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.NumericWidenOp GHC.Int.Int32 GHC.Types.Float
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.NumericWidenOp GHC.Int.Int64 GHC.Types.Double
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.NumericWidenOp GHC.Int.Int64 GHC.Types.Float
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.NumericWidenOp GHC.Types.Double GHC.Int.Int32
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.NumericWidenOp GHC.Types.Double GHC.Int.Int64
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.NumericWidenOp GHC.Types.Double GHC.Types.Float
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.NumericWidenOp GHC.Types.Double GHC.Types.Int
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.NumericWidenOp GHC.Types.Float GHC.Int.Int32
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.NumericWidenOp GHC.Types.Float GHC.Int.Int64
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.NumericWidenOp GHC.Types.Float GHC.Types.Double
- DataFrame.Internal.Nullable: instance DataFrame.Internal.Nullable.NumericWidenOp GHC.Types.Int GHC.Types.Double
- DataFrame.Internal.Nullable: nullArithOp :: NullableArithOp a b c => (BaseType a -> BaseType a -> BaseType a) -> a -> b -> c
- DataFrame.Internal.Nullable: nullCmpOp :: NullableCmpOp a b e => (BaseType a -> BaseType a -> Bool) -> a -> b -> e
- DataFrame.Internal.Nullable: type WidenResult a b = NullLift2Result a b Promote BaseType a BaseType b
- DataFrame.Internal.Nullable: type WidenResultDiv a b = NullLift2Result a b PromoteDiv BaseType a BaseType b
- DataFrame.Internal.Nullable: type family NullCmpResult a b
- DataFrame.Internal.Nullable: widen1 :: NumericWidenOp a b => a -> Promote a b
- DataFrame.Internal.Nullable: widen2 :: NumericWidenOp a b => b -> Promote a b
- DataFrame.Internal.Nullable: widenArithOp :: NumericWidenOp a b => (Promote a b -> Promote a b -> Promote a b) -> a -> b -> Promote a b
- DataFrame.Internal.Nullable: widenCmpOp :: NumericWidenOp a b => (Promote a b -> Promote a b -> Bool) -> a -> b -> Bool
- DataFrame.Internal.PackedText: Offs32 :: {-# UNPACK #-} !Vector Int32 -> PackedOffsets
- DataFrame.Internal.PackedText: Offs64 :: {-# UNPACK #-} !Vector Int -> PackedOffsets
- DataFrame.Internal.PackedText: PackedTextData :: {-# UNPACK #-} !Array -> !PackedOffsets -> !Maybe PackedSel -> !Bool -> PackedTextData
- DataFrame.Internal.PackedText: Sel32 :: {-# UNPACK #-} !Vector Int32 -> PackedSel
- DataFrame.Internal.PackedText: Sel64 :: {-# UNPACK #-} !Vector Int -> PackedSel
- DataFrame.Internal.PackedText: [ptBytes] :: PackedTextData -> {-# UNPACK #-} !Array
- DataFrame.Internal.PackedText: [ptCanonicalSel] :: PackedTextData -> !Bool
- DataFrame.Internal.PackedText: [ptOffsets] :: PackedTextData -> !PackedOffsets
- DataFrame.Internal.PackedText: [ptSel] :: PackedTextData -> !Maybe PackedSel
- DataFrame.Internal.PackedText: data PackedOffsets
- DataFrame.Internal.PackedText: data PackedSel
- DataFrame.Internal.PackedText: data PackedTextData
- DataFrame.Internal.PackedText: mkOffsets :: Vector Int -> PackedOffsets
- DataFrame.Internal.PackedText: mkPackedContiguous :: Array -> Vector Int -> PackedTextData
- DataFrame.Internal.PackedText: mkPackedContiguous32 :: Array -> Vector Int32 -> PackedTextData
- DataFrame.Internal.PackedText: mkSel :: Int -> Vector Int -> PackedSel
- DataFrame.Internal.PackedText: offAt :: PackedOffsets -> Int -> Int
- DataFrame.Internal.PackedText: offCount :: PackedOffsets -> Int
- DataFrame.Internal.PackedText: packedGather :: Vector Int -> PackedTextData -> PackedTextData
- DataFrame.Internal.PackedText: packedIndexText :: PackedTextData -> Int -> Text
- DataFrame.Internal.PackedText: packedLength :: PackedTextData -> Int
- DataFrame.Internal.PackedText: packedRowOffsets :: PackedTextData -> Maybe (Array, PackedOffsets)
- DataFrame.Internal.PackedText: packedSlice :: PackedTextData -> Int -> (Array, Int, Int)
- DataFrame.Internal.PackedText: packedTake :: Int -> PackedTextData -> PackedTextData
- DataFrame.Internal.PackedText: selAt :: PackedSel -> Int -> Int
- DataFrame.Internal.PackedText: selLength :: PackedSel -> Int
- DataFrame.Internal.PackedText: sliceCmpBytes :: Array -> Int -> Int -> Array -> Int -> Int -> Ordering
- DataFrame.Internal.PackedText: sliceEqBytes :: Array -> Int -> Int -> Array -> Int -> Int -> Bool
- DataFrame.Internal.ParRadixSort: parSortByHash :: Int -> Vector Int -> (Vector Int, Vector Int)
- DataFrame.Internal.ParRadixSort: parSortThreshold :: Int
- DataFrame.Internal.Pretty: (<+>) :: Doc -> Doc -> Doc
- DataFrame.Internal.Pretty: data Doc
- DataFrame.Internal.Pretty: defaultWidth :: Int
- DataFrame.Internal.Pretty: group :: Doc -> Doc
- DataFrame.Internal.Pretty: hardline :: Doc
- DataFrame.Internal.Pretty: hcat :: [Doc] -> Doc
- DataFrame.Internal.Pretty: infixr 6 <+>
- DataFrame.Internal.Pretty: instance GHC.Base.Monoid DataFrame.Internal.Pretty.Doc
- DataFrame.Internal.Pretty: instance GHC.Base.Semigroup DataFrame.Internal.Pretty.Doc
- DataFrame.Internal.Pretty: line :: Doc
- DataFrame.Internal.Pretty: nest :: Int -> Doc -> Doc
- DataFrame.Internal.Pretty: parens :: Doc -> Doc
- DataFrame.Internal.Pretty: parensWhenBroken :: Doc -> Doc
- DataFrame.Internal.Pretty: punctuate :: Doc -> [Doc] -> [Doc]
- DataFrame.Internal.Pretty: render :: Int -> Doc -> String
- DataFrame.Internal.Pretty: text :: String -> Doc
- DataFrame.Internal.RadixRank: rankByHash :: PrimMonad m => (Int -> m Int) -> Int -> m (Vector Int)
- DataFrame.Internal.RadixRank: sortKey :: Int -> Int
- DataFrame.Internal.RowHash: computeRowHashesIO :: Int -> [Column] -> IO (Vector Int)
- DataFrame.Internal.RowHash: hashRowRange :: IOVector Int -> Int -> Int -> [Column] -> IO ()
- DataFrame.Internal.RowHash: parRowHashThreshold :: Int
- DataFrame.Internal.Simplify: data PredFact
- DataFrame.Internal.Simplify: entails :: [PredFact] -> Expr Bool -> Maybe Bool
- DataFrame.Internal.Simplify: factFalse :: Expr Bool -> Maybe PredFact
- DataFrame.Internal.Simplify: factTrue :: Expr Bool -> Maybe PredFact
- DataFrame.Internal.Simplify: instance GHC.Classes.Eq DataFrame.Internal.Simplify.Cmp
- DataFrame.Internal.Simplify: instance GHC.Classes.Eq DataFrame.Internal.Simplify.NullK
- DataFrame.Internal.Simplify: simplify :: Columnable a => Expr a -> Expr a
- DataFrame.Internal.Simplify: simplifyPredicatePair :: Columnable a => Bool -> Expr a -> Expr a -> Maybe (Expr a)
- DataFrame.Internal.Types: RBoxed :: Rep
- DataFrame.Internal.Types: RNullableBoxed :: Rep
- DataFrame.Internal.Types: RUnboxed :: Rep
- DataFrame.Internal.Types: That :: b -> These a b
- DataFrame.Internal.Types: These :: a -> b -> These a b
- DataFrame.Internal.Types: This :: a -> These a b
- DataFrame.Internal.Types: [SFalse] :: SBool 'False
- DataFrame.Internal.Types: [STrue] :: SBool 'True
- DataFrame.Internal.Types: class SBoolI (b :: Bool)
- DataFrame.Internal.Types: data Rep
- DataFrame.Internal.Types: data SBool (b :: Bool)
- DataFrame.Internal.Types: data These a b
- DataFrame.Internal.Types: instance (GHC.Classes.Eq a, GHC.Classes.Eq b) => GHC.Classes.Eq (DataFrame.Internal.Types.These a b)
- DataFrame.Internal.Types: instance (GHC.Classes.Ord a, GHC.Classes.Ord b) => GHC.Classes.Ord (DataFrame.Internal.Types.These a b)
- DataFrame.Internal.Types: instance (GHC.Read.Read a, GHC.Read.Read b) => GHC.Read.Read (DataFrame.Internal.Types.These a b)
- DataFrame.Internal.Types: instance (GHC.Show.Show a, GHC.Show.Show b) => GHC.Show.Show (DataFrame.Internal.Types.These a b)
- DataFrame.Internal.Types: instance Data.Foldable.Foldable (DataFrame.Internal.Types.These a)
- DataFrame.Internal.Types: instance Data.Traversable.Traversable (DataFrame.Internal.Types.These a)
- DataFrame.Internal.Types: instance DataFrame.Internal.Types.SBoolI 'GHC.Types.False
- DataFrame.Internal.Types: instance DataFrame.Internal.Types.SBoolI 'GHC.Types.True
- DataFrame.Internal.Types: instance GHC.Base.Functor (DataFrame.Internal.Types.These a)
- DataFrame.Internal.Types: sFloating :: SBoolI (FloatingTypes a) => SBool (FloatingTypes a)
- DataFrame.Internal.Types: sIntegral :: SBoolI (IntegralTypes a) => SBool (IntegralTypes a)
- DataFrame.Internal.Types: sNumeric :: SBoolI (Numeric a) => SBool (Numeric a)
- DataFrame.Internal.Types: sUnbox :: SBoolI (Unboxable a) => SBool (Unboxable a)
- DataFrame.Internal.Types: sbool :: SBoolI b => SBool b
- DataFrame.Internal.Types: type Columnable' a = (Typeable a, Show a, Eq a)
- DataFrame.Internal.Types: type FloatingIf a = When FloatingTypes a (Real a, Fractional a)
- DataFrame.Internal.Types: type IntegralIf a = When IntegralTypes a Integral a
- DataFrame.Internal.Types: type UnboxIf a = When Unboxable a Unbox a
- DataFrame.Internal.Types: type family PromoteDiv a b
- DataFrame.Internal.Utf8: isUtf8Boundary :: Word8 -> Bool
- DataFrame.Internal.Utf8: isValidUtf8Slice :: Array -> Int -> Int -> Bool
- DataFrame.Internal.Utf8: lenientDecodeSlice :: Array -> Int -> Int -> Text
- DataFrame.Internal.Utf8: sliceTextVector :: Array -> Vector Int -> Vector Text
- DataFrame.Operators: (.&&) :: (NullableCmpOp a b (NullCmpResult a b), BaseType a ~ Bool) => Expr a -> Expr b -> Expr (NullCmpResult a b)
- DataFrame.Operators: (.&&.) :: Expr Bool -> Expr Bool -> Expr Bool
- DataFrame.Operators: (.*) :: (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b (Promote (BaseType a) (BaseType b)) (WidenResult a b), Num (Promote (BaseType a) (BaseType b))) => Expr a -> Expr b -> Expr (WidenResult a b)
- DataFrame.Operators: (.*.) :: (Columnable a, Num a) => Expr a -> Expr a -> Expr a
- DataFrame.Operators: (.+) :: (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b (Promote (BaseType a) (BaseType b)) (WidenResult a b), Num (Promote (BaseType a) (BaseType b))) => Expr a -> Expr b -> Expr (WidenResult a b)
- DataFrame.Operators: (.+.) :: (Columnable a, Num a) => Expr a -> Expr a -> Expr a
- DataFrame.Operators: (.-) :: (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b (Promote (BaseType a) (BaseType b)) (WidenResult a b), Num (Promote (BaseType a) (BaseType b))) => Expr a -> Expr b -> Expr (WidenResult a b)
- DataFrame.Operators: (.-.) :: (Columnable a, Num a) => Expr a -> Expr a -> Expr a
- DataFrame.Operators: (./) :: (DivWidenOp (BaseType a) (BaseType b), NullLift2Op a b (PromoteDiv (BaseType a) (BaseType b)) (WidenResultDiv a b), Fractional (PromoteDiv (BaseType a) (BaseType b))) => Expr a -> Expr b -> Expr (WidenResultDiv a b)
- DataFrame.Operators: (./.) :: (Columnable a, Fractional a) => Expr a -> Expr a -> Expr a
- DataFrame.Operators: (./=) :: (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Eq (Promote (BaseType a) (BaseType b))) => Expr a -> Expr b -> Expr (NullCmpResult a b)
- DataFrame.Operators: (./=.) :: (Columnable a, Eq a) => Expr a -> Expr a -> Expr Bool
- DataFrame.Operators: (.<) :: (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Ord (Promote (BaseType a) (BaseType b))) => Expr a -> Expr b -> Expr (NullCmpResult a b)
- DataFrame.Operators: (.<.) :: (Columnable a, Ord a) => Expr a -> Expr a -> Expr Bool
- DataFrame.Operators: (.<=) :: (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Ord (Promote (BaseType a) (BaseType b))) => Expr a -> Expr b -> Expr (NullCmpResult a b)
- DataFrame.Operators: (.<=.) :: (Columnable a, Ord a) => Expr a -> Expr a -> Expr Bool
- DataFrame.Operators: (.=) :: Columnable a => Text -> Expr a -> NamedExpr
- DataFrame.Operators: (.==) :: (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Eq (Promote (BaseType a) (BaseType b))) => Expr a -> Expr b -> Expr (NullCmpResult a b)
- DataFrame.Operators: (.==.) :: (Columnable a, Eq a) => Expr a -> Expr a -> Expr Bool
- DataFrame.Operators: (.>) :: (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Ord (Promote (BaseType a) (BaseType b))) => Expr a -> Expr b -> Expr (NullCmpResult a b)
- DataFrame.Operators: (.>.) :: (Columnable a, Ord a) => Expr a -> Expr a -> Expr Bool
- DataFrame.Operators: (.>=) :: (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Ord (Promote (BaseType a) (BaseType b))) => Expr a -> Expr b -> Expr (NullCmpResult a b)
- DataFrame.Operators: (.>=.) :: (Columnable a, Ord a) => Expr a -> Expr a -> Expr Bool
- DataFrame.Operators: (.^) :: (Columnable (BaseType a), Columnable (BaseType b), Num (BaseType a), Integral (BaseType b), NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b (BaseType a) a, Num (Promote (BaseType a) (BaseType b))) => Expr a -> Expr b -> Expr a
- DataFrame.Operators: (.^.) :: (Columnable a, Columnable b, Num a, Integral b) => Expr a -> Expr b -> Expr a
- DataFrame.Operators: (.^^) :: (Columnable (BaseType a), Columnable (BaseType b), Fractional (BaseType a), Integral (BaseType b), NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b (BaseType a) a, Num (Promote (BaseType a) (BaseType b))) => Expr a -> Expr b -> Expr a
- DataFrame.Operators: (.^^.) :: (Columnable a, Columnable b, Fractional a, Integral b) => Expr a -> Expr b -> Expr a
- DataFrame.Operators: (.||) :: (NullableCmpOp a b (NullCmpResult a b), BaseType a ~ Bool) => Expr a -> Expr b -> Expr (NullCmpResult a b)
- DataFrame.Operators: (.||.) :: Expr Bool -> Expr Bool -> Expr Bool
- DataFrame.Operators: (|>) :: a -> (a -> b) -> b
- DataFrame.Operators: [NullAnd] :: forall a b. (NullableCmpOp a b (NullCmpResult a b), BaseType a ~ Bool) => NullAnd a b (NullCmpResult a b)
- DataFrame.Operators: [NullEq] :: forall a b. (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Eq (Promote (BaseType a) (BaseType b))) => NullEq a b (NullCmpResult a b)
- DataFrame.Operators: [NullGeq] :: forall a b. (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Ord (Promote (BaseType a) (BaseType b))) => NullGeq a b (NullCmpResult a b)
- DataFrame.Operators: [NullGt] :: forall a b. (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Ord (Promote (BaseType a) (BaseType b))) => NullGt a b (NullCmpResult a b)
- DataFrame.Operators: [NullLeq] :: forall a b. (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Ord (Promote (BaseType a) (BaseType b))) => NullLeq a b (NullCmpResult a b)
- DataFrame.Operators: [NullLt] :: forall a b. (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Ord (Promote (BaseType a) (BaseType b))) => NullLt a b (NullCmpResult a b)
- DataFrame.Operators: [NullNeq] :: forall a b. (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Eq (Promote (BaseType a) (BaseType b))) => NullNeq a b (NullCmpResult a b)
- DataFrame.Operators: [NullOr] :: forall a b. (NullableCmpOp a b (NullCmpResult a b), BaseType a ~ Bool) => NullOr a b (NullCmpResult a b)
- DataFrame.Operators: as :: Columnable a => Expr a -> Text -> NamedExpr
- DataFrame.Operators: col :: Columnable a => Text -> Expr a
- DataFrame.Operators: data NullAnd a b c
- DataFrame.Operators: data NullEq a b c
- DataFrame.Operators: data NullGeq a b c
- DataFrame.Operators: data NullGt a b c
- DataFrame.Operators: data NullLeq a b c
- DataFrame.Operators: data NullLt a b c
- DataFrame.Operators: data NullNeq a b c
- DataFrame.Operators: data NullOr a b c
- DataFrame.Operators: ifThenElse :: Columnable a => Expr Bool -> Expr a -> Expr a -> Expr a
- DataFrame.Operators: infix 4 .>=
- DataFrame.Operators: infixl 6 .-
- DataFrame.Operators: infixl 7 ./
- DataFrame.Operators: infixr 0 .=
- DataFrame.Operators: infixr 2 .||
- DataFrame.Operators: infixr 3 .&&
- DataFrame.Operators: infixr 8 .^.
- DataFrame.Operators: instance DataFrame.Internal.Expression.BinaryOp DataFrame.Operators.NullAnd
- DataFrame.Operators: instance DataFrame.Internal.Expression.BinaryOp DataFrame.Operators.NullEq
- DataFrame.Operators: instance DataFrame.Internal.Expression.BinaryOp DataFrame.Operators.NullGeq
- DataFrame.Operators: instance DataFrame.Internal.Expression.BinaryOp DataFrame.Operators.NullGt
- DataFrame.Operators: instance DataFrame.Internal.Expression.BinaryOp DataFrame.Operators.NullLeq
- DataFrame.Operators: instance DataFrame.Internal.Expression.BinaryOp DataFrame.Operators.NullLt
- DataFrame.Operators: instance DataFrame.Internal.Expression.BinaryOp DataFrame.Operators.NullNeq
- DataFrame.Operators: instance DataFrame.Internal.Expression.BinaryOp DataFrame.Operators.NullOr
- DataFrame.Operators: lift2Decorated :: (Columnable c, Columnable b, Columnable a) => (c -> b -> a) -> Text -> Maybe Text -> Bool -> Int -> Expr c -> Expr b -> Expr a
- DataFrame.Operators: liftDecorated :: (Columnable a, Columnable b) => (a -> b) -> Text -> Maybe Text -> Expr a -> Expr b
- DataFrame.Operators: lit :: Columnable a => a -> Expr a
- DataFrame.Operators: name :: Show a => Expr a -> Text
- DataFrame.Typed.Generic: instance (GHC.TypeLits.KnownSymbol name, DataFrame.Internal.Column.Columnable a) => DataFrame.Typed.Generic.GHasColumns (GHC.Generics.M1 GHC.Generics.S ('GHC.Generics.MetaSel ('GHC.Maybe.Just name) su ss ds) (GHC.Generics.K1 i a))
- DataFrame.Typed.Schema: instance (GHC.TypeLits.KnownSymbol name, Data.Typeable.Internal.Typeable a, DataFrame.Internal.Column.Columnable a, DataFrame.Typed.Schema.KnownSchema rest) => DataFrame.Typed.Schema.KnownSchema ('(name, a) : rest)
+ DataFrame.Errors: [ExpectedNonNullableException] :: DataFrameException
+ DataFrame.Internal.Aggregation.Kernel.Dense: denseMaxMinusMin :: Vector Int -> Int -> Column -> Column -> Maybe Column
+ DataFrame.Internal.Aggregation.Kernel.Dense: denseReduce :: Reduction -> Vector Int -> Int -> Column -> Maybe Column
+ DataFrame.Internal.Aggregation.Kernel.Fused: data FusedAgg
+ DataFrame.Internal.Aggregation.Kernel.Fused: data GatherAgg
+ DataFrame.Internal.Aggregation.Kernel.Fused: mkFusedAgg :: Int -> Vector Int -> Reduction -> Column -> Maybe FusedAgg
+ DataFrame.Internal.Aggregation.Kernel.Fused: mkGatherAgg :: Int -> Vector Int -> Vector Int -> Reduction -> Column -> Maybe GatherAgg
+ DataFrame.Internal.Aggregation.Kernel.Fused: runFusedAggs :: Int -> Int -> [FusedAgg] -> [Column]
+ DataFrame.Internal.Aggregation.Kernel.Fused: runGatherAggs :: Vector Int -> Vector Int -> Int -> [GatherAgg] -> [Column]
+ DataFrame.Internal.Aggregation.Kernel.Moments: Moments :: Column -> Column -> Column -> Column -> Column -> Column -> Moments
+ DataFrame.Internal.Aggregation.Kernel.Moments: [mN] :: Moments -> Column
+ DataFrame.Internal.Aggregation.Kernel.Moments: [mSx] :: Moments -> Column
+ DataFrame.Internal.Aggregation.Kernel.Moments: [mSxx] :: Moments -> Column
+ DataFrame.Internal.Aggregation.Kernel.Moments: [mSxy] :: Moments -> Column
+ DataFrame.Internal.Aggregation.Kernel.Moments: [mSy] :: Moments -> Column
+ DataFrame.Internal.Aggregation.Kernel.Moments: [mSyy] :: Moments -> Column
+ DataFrame.Internal.Aggregation.Kernel.Moments: data Moments
+ DataFrame.Internal.Aggregation.Kernel.Moments: momentScatter :: Vector Int -> Int -> Column -> Column -> Maybe Moments
+ DataFrame.Internal.Aggregation.Kernel.Moments: momentScatterPar :: Vector Int -> Vector Int -> Int -> Column -> Column -> Maybe Moments
+ DataFrame.Internal.Aggregation.Kernel.Moments: momentStreamPar :: Vector Int -> Int -> Column -> Column -> Maybe Moments
+ DataFrame.Internal.Aggregation.Kernel.Scatter: groupRangeBounds :: Vector Int -> Int -> Int -> Vector Int
+ DataFrame.Internal.Aggregation.Kernel.Scatter: groupSlices :: Int -> [(Int, Int)]
+ DataFrame.Internal.Aggregation.Kernel.Scatter: maxMinusMinScatterPar :: Vector Int -> Vector Int -> Int -> Column -> Column -> Maybe Column
+ DataFrame.Internal.Aggregation.Kernel.Scatter: overGroupsAcc :: Vector Int -> Vector Int -> Int -> Int -> acc -> (acc -> Int -> acc) -> (Int -> acc -> IO ()) -> IO ()
+ DataFrame.Internal.Aggregation.Kernel.Scatter: rtgFromVis :: Vector Int -> Vector Int -> Int -> Vector Int
+ DataFrame.Internal.Aggregation.Kernel.Scatter: scatterReduce :: Reduction -> Vector Int -> Int -> Column -> Maybe Column
+ DataFrame.Internal.Aggregation.Kernel.Scatter: scatterReducePar :: Reduction -> Vector Int -> Vector Int -> Int -> Column -> Maybe Column
+ DataFrame.Internal.Aggregation.Kernel.Scatter: streamGroupCap :: Int
+ DataFrame.Internal.Aggregation.Kernel.Scatter: top2SndScatter :: (Unbox a, Real a) => Vector Int -> Int -> Vector a -> Vector Double
+ DataFrame.Internal.Aggregation.Plan: MomentPlan :: Text -> Text -> Text -> Text -> Text -> Text -> Text -> Text -> MomentPlan
+ DataFrame.Internal.Aggregation.Plan: PlanMaxMinusMin :: Text -> Text -> AggPlan
+ DataFrame.Internal.Aggregation.Plan: PlanMedian :: Text -> AggPlan
+ DataFrame.Internal.Aggregation.Plan: PlanScatter :: Reduction -> Text -> AggPlan
+ DataFrame.Internal.Aggregation.Plan: [mpColX] :: MomentPlan -> Text
+ DataFrame.Internal.Aggregation.Plan: [mpColY] :: MomentPlan -> Text
+ DataFrame.Internal.Aggregation.Plan: [mpNName] :: MomentPlan -> Text
+ DataFrame.Internal.Aggregation.Plan: [mpSxName] :: MomentPlan -> Text
+ DataFrame.Internal.Aggregation.Plan: [mpSxxName] :: MomentPlan -> Text
+ DataFrame.Internal.Aggregation.Plan: [mpSxyName] :: MomentPlan -> Text
+ DataFrame.Internal.Aggregation.Plan: [mpSyName] :: MomentPlan -> Text
+ DataFrame.Internal.Aggregation.Plan: [mpSyyName] :: MomentPlan -> Text
+ DataFrame.Internal.Aggregation.Plan: data AggPlan
+ DataFrame.Internal.Aggregation.Plan: data MomentPlan
+ DataFrame.Internal.Aggregation.Plan: instance GHC.Classes.Eq DataFrame.Internal.Aggregation.Plan.Role
+ DataFrame.Internal.Aggregation.Plan: instance GHC.Classes.Eq DataFrame.Internal.Aggregation.Plan.Term
+ DataFrame.Internal.Aggregation.Plan: instance GHC.Classes.Ord DataFrame.Internal.Aggregation.Plan.Role
+ DataFrame.Internal.Aggregation.Plan: instance GHC.Classes.Ord DataFrame.Internal.Aggregation.Plan.Term
+ DataFrame.Internal.Aggregation.Plan: instance GHC.Show.Show DataFrame.Internal.Aggregation.Plan.Role
+ DataFrame.Internal.Aggregation.Plan: instance GHC.Show.Show DataFrame.Internal.Aggregation.Plan.Term
+ DataFrame.Internal.Aggregation.Plan: planAgg :: GroupedDataFrame -> UExpr -> Maybe AggPlan
+ DataFrame.Internal.Aggregation.Plan: planMoments :: GroupedDataFrame -> [(Text, UExpr)] -> Maybe MomentPlan
+ DataFrame.Internal.Aggregation.Reduction: RCount :: Reduction
+ DataFrame.Internal.Aggregation.Reduction: RMax :: Reduction
+ DataFrame.Internal.Aggregation.Reduction: RMean :: Reduction
+ DataFrame.Internal.Aggregation.Reduction: RMin :: Reduction
+ DataFrame.Internal.Aggregation.Reduction: RStd :: Reduction
+ DataFrame.Internal.Aggregation.Reduction: RSum :: Reduction
+ DataFrame.Internal.Aggregation.Reduction: RTop2Snd :: Reduction
+ DataFrame.Internal.Aggregation.Reduction: RTop2Sum :: Reduction
+ DataFrame.Internal.Aggregation.Reduction: RVar :: Reduction
+ DataFrame.Internal.Aggregation.Reduction: cleanDoubleVector :: Column -> Maybe (Vector Double)
+ DataFrame.Internal.Aggregation.Reduction: data Reduction
+ DataFrame.Internal.Aggregation.Reduction: instance GHC.Classes.Eq DataFrame.Internal.Aggregation.Reduction.Reduction
+ DataFrame.Internal.Aggregation.Reduction: instance GHC.Show.Show DataFrame.Internal.Aggregation.Reduction.Reduction
+ DataFrame.Internal.Algorithms.Hash: fnvOffset :: Int
+ DataFrame.Internal.Algorithms.Hash: mixBool :: Int -> Bool -> Int
+ DataFrame.Internal.Algorithms.Hash: mixBytes :: Int -> Array -> Int -> Int -> Int
+ DataFrame.Internal.Algorithms.Hash: mixChar :: Int -> Char -> Int
+ DataFrame.Internal.Algorithms.Hash: mixDouble :: Int -> Double -> Int
+ DataFrame.Internal.Algorithms.Hash: mixInt :: Int -> Int -> Int
+ DataFrame.Internal.Algorithms.Hash: mixShow :: Show a => Int -> a -> Int
+ DataFrame.Internal.Algorithms.Hash: mixText :: Int -> Text -> Int
+ DataFrame.Internal.Algorithms.Hash: nullSalt :: Int
+ DataFrame.Internal.Algorithms.Rank.Radix: rankByHash :: PrimMonad m => (Int -> m Int) -> Int -> m (Vector Int)
+ DataFrame.Internal.Algorithms.Rank.Radix: sortKey :: Int -> Int
+ DataFrame.Internal.Algorithms.Sort.Radix.Parallel: parSortByHash :: Int -> Vector Int -> (Vector Int, Vector Int)
+ DataFrame.Internal.Algorithms.Sort.Radix.Parallel: parSortThreshold :: Int
+ DataFrame.Internal.Column.Base: [BoxedColumn] :: forall a. Columnable a => Maybe Bitmap -> Vector a -> Column
+ DataFrame.Internal.Column.Base: [MBoxedColumn] :: forall a. Columnable a => IOVector a -> MutableColumn
+ DataFrame.Internal.Column.Base: [MUnboxedColumn] :: forall a. (Columnable a, Unbox a) => IOVector a -> MutableColumn
+ DataFrame.Internal.Column.Base: [MergedColumn] :: !Column -> !Column -> Column
+ DataFrame.Internal.Column.Base: [PackedText] :: Maybe Bitmap -> {-# UNPACK #-} !PackedTextData -> Column
+ DataFrame.Internal.Column.Base: [TColumn] :: forall a. Columnable a => Column -> TypedColumn a
+ DataFrame.Internal.Column.Base: [UnboxedColumn] :: forall a. (Columnable a, Unbox a) => Maybe Bitmap -> Vector a -> Column
+ DataFrame.Internal.Column.Base: class ColumnifyRep (r :: Rep) a
+ DataFrame.Internal.Column.Base: columnBitmap :: Column -> Maybe Bitmap
+ DataFrame.Internal.Column.Base: columnLength :: Column -> Int
+ DataFrame.Internal.Column.Base: data Column
+ DataFrame.Internal.Column.Base: data MutableColumn
+ DataFrame.Internal.Column.Base: data TypedColumn a
+ DataFrame.Internal.Column.Base: eqBoxedCols :: Eq a => Maybe Bitmap -> Vector a -> Maybe Bitmap -> Vector a -> Bool
+ DataFrame.Internal.Column.Base: eqPackedCols :: Maybe Bitmap -> PackedTextData -> Maybe Bitmap -> PackedTextData -> Bool
+ DataFrame.Internal.Column.Base: instance (DataFrame.Internal.Column.Base.Columnable a, Data.Vector.Unboxed.Base.Unbox a) => DataFrame.Internal.Column.Base.ColumnifyRep 'DataFrame.Internal.Column.Types.RUnboxed a
+ DataFrame.Internal.Column.Base: instance DataFrame.Internal.Column.Base.Columnable a => DataFrame.Internal.Column.Base.ColumnifyRep 'DataFrame.Internal.Column.Types.RBoxed a
+ DataFrame.Internal.Column.Base: instance DataFrame.Internal.Column.Base.Columnable a => DataFrame.Internal.Column.Base.ColumnifyRep 'DataFrame.Internal.Column.Types.RNullableBoxed (GHC.Maybe.Maybe a)
+ DataFrame.Internal.Column.Base: instance GHC.Classes.Eq DataFrame.Internal.Column.Base.Column
+ DataFrame.Internal.Column.Base: instance GHC.Classes.Eq a => GHC.Classes.Eq (DataFrame.Internal.Column.Base.TypedColumn a)
+ DataFrame.Internal.Column.Base: instance GHC.Show.Show DataFrame.Internal.Column.Base.Column
+ DataFrame.Internal.Column.Base: instance GHC.Show.Show a => GHC.Show.Show (DataFrame.Internal.Column.Base.TypedColumn a)
+ DataFrame.Internal.Column.Base: materializeMerged :: Column -> Column
+ DataFrame.Internal.Column.Base: materializePacked :: Column -> Column
+ DataFrame.Internal.Column.Base: mergeEager :: Column -> Column -> Column
+ DataFrame.Internal.Column.Base: mergedHead :: Column -> Column
+ DataFrame.Internal.Column.Base: mkMergedColumns :: Column -> Column -> Column
+ DataFrame.Internal.Column.Base: takeColumn :: Int -> Column -> Column
+ DataFrame.Internal.Column.Base: toColumnRep :: ColumnifyRep r a => Vector a -> Column
+ DataFrame.Internal.Column.Base: type Columnable a = (Columnable' a, ColumnifyRep KindOf a a, UnboxIf a, IntegralIf a, FloatingIf a, SBoolI Unboxable a, SBoolI Numeric a, SBoolI IntegralTypes a, SBoolI FloatingTypes a)
+ DataFrame.Internal.Column.Base: unwrapTypedColumn :: TypedColumn a -> Column
+ DataFrame.Internal.Column.Bitmap: Validity :: !Maybe Bitmap -> {-# UNPACK #-} !Int -> Validity
+ DataFrame.Internal.Column.Bitmap: allValidBitmap :: Int -> Bitmap
+ DataFrame.Internal.Column.Bitmap: allValidBitmap' :: Int -> ST s (MVector s Word8)
+ DataFrame.Internal.Column.Bitmap: andBitmaps :: Bitmap -> Bitmap -> Bitmap
+ DataFrame.Internal.Column.Bitmap: bitmapConcat :: Int -> Bitmap -> Int -> Bitmap -> Bitmap
+ DataFrame.Internal.Column.Bitmap: bitmapSlice :: Int -> Int -> Bitmap -> Bitmap
+ DataFrame.Internal.Column.Bitmap: bitmapTestBit :: Bitmap -> Int -> Bool
+ DataFrame.Internal.Column.Bitmap: buildBitmapFromNulls :: Int -> [Int] -> Bitmap
+ DataFrame.Internal.Column.Bitmap: buildBitmapFromNulls' :: Int -> Vector Int -> Vector Word8
+ DataFrame.Internal.Column.Bitmap: buildBitmapFromValid :: Vector Word8 -> Bitmap
+ DataFrame.Internal.Column.Bitmap: concatValidity :: [Validity] -> Maybe Bitmap
+ DataFrame.Internal.Column.Bitmap: data Validity
+ DataFrame.Internal.Column.Bitmap: packValidity :: Int -> MVector s Word8 -> ST s Bitmap
+ DataFrame.Internal.Column.Bitmap: popCountUpTo :: Int -> Bitmap -> Int
+ DataFrame.Internal.Column.Bitmap: type Bitmap = Vector Word8
+ DataFrame.Internal.Column.Bitmap: vBitmap :: Validity -> Maybe Bitmap
+ DataFrame.Internal.Column.Bitmap: vRowCount :: Validity -> Int
+ DataFrame.Internal.Column.Builder: TextChunk :: !Array -> !Int -> !Vector Int -> !Maybe Bitmap -> TextChunk
+ DataFrame.Internal.Column.Builder: [tcBitmap] :: TextChunk -> !Maybe Bitmap
+ DataFrame.Internal.Column.Builder: [tcBytes] :: TextChunk -> !Array
+ DataFrame.Internal.Column.Builder: [tcOffsets] :: TextChunk -> !Vector Int
+ DataFrame.Internal.Column.Builder: [tcUsed] :: TextChunk -> !Int
+ DataFrame.Internal.Column.Builder: appendDouble :: DoubleBuilder s -> Double -> ST s ()
+ DataFrame.Internal.Column.Builder: appendInt :: IntBuilder s -> Int -> ST s ()
+ DataFrame.Internal.Column.Builder: appendNull :: ColumnBuilder b => b s -> ST s ()
+ DataFrame.Internal.Column.Builder: appendNum :: Unbox a => NumBuilder a s -> a -> ST s ()
+ DataFrame.Internal.Column.Builder: appendText :: TextBuilder s -> Text -> ST s ()
+ DataFrame.Internal.Column.Builder: appendTextSlice :: TextBuilder s -> Array -> Int -> Int -> ST s ()
+ DataFrame.Internal.Column.Builder: appendTextSliceFromPtr :: TextBuilder s -> Ptr Word8 -> Int -> ST s ()
+ DataFrame.Internal.Column.Builder: builderLength :: ColumnBuilder b => b s -> ST s Int
+ DataFrame.Internal.Column.Builder: class ColumnBuilder (b :: Type -> Type)
+ DataFrame.Internal.Column.Builder: concatColumns :: [Column] -> Column
+ DataFrame.Internal.Column.Builder: data NumBuilder a s
+ DataFrame.Internal.Column.Builder: data TextBuilder s
+ DataFrame.Internal.Column.Builder: data TextChunk
+ DataFrame.Internal.Column.Builder: freezeBuilder :: ColumnBuilder b => b s -> ST s Column
+ DataFrame.Internal.Column.Builder: freezeTextChunk :: TextBuilder s -> ST s TextChunk
+ DataFrame.Internal.Column.Builder: instance (DataFrame.Internal.Column.Base.Columnable a, Data.Vector.Unboxed.Base.Unbox a) => DataFrame.Internal.Column.Builder.ColumnBuilder (DataFrame.Internal.Column.Builder.NumBuilder a)
+ DataFrame.Internal.Column.Builder: instance DataFrame.Internal.Column.Builder.ColumnBuilder DataFrame.Internal.Column.Builder.TextBuilder
+ DataFrame.Internal.Column.Builder: mergeTextChunks :: [TextChunk] -> Column
+ DataFrame.Internal.Column.Builder: newDoubleBuilder :: Int -> ST s (DoubleBuilder s)
+ DataFrame.Internal.Column.Builder: newIntBuilder :: Int -> ST s (IntBuilder s)
+ DataFrame.Internal.Column.Builder: newNumBuilder :: Unbox a => a -> Int -> ST s (NumBuilder a s)
+ DataFrame.Internal.Column.Builder: newTextBuilder :: Int -> Int -> ST s (TextBuilder s)
+ DataFrame.Internal.Column.Builder: type DoubleBuilder = NumBuilder Double
+ DataFrame.Internal.Column.Builder: type IntBuilder = NumBuilder Int
+ DataFrame.Internal.Column.Conversion: columnToTextVec :: Column -> Vector Text
+ DataFrame.Internal.Column.Conversion: finalizeParseResult :: Unbox a => STVector s a -> STVector s Word8 -> Bool -> ST s (Maybe (Maybe Bitmap, Vector a))
+ DataFrame.Internal.Column.Conversion: fromList :: (Columnable a, ColumnifyRep (KindOf a) a) => [a] -> Column
+ DataFrame.Internal.Column.Conversion: fromUnboxedVector :: (Columnable a, Unbox a) => Vector a -> Column
+ DataFrame.Internal.Column.Conversion: fromVector :: (Columnable a, ColumnifyRep (KindOf a) a) => Vector a -> Column
+ DataFrame.Internal.Column.Conversion: mkRandom :: (RandomGen g, Columnable a, ColumnifyRep (KindOf a) a, UniformRange a) => g -> Int -> a -> a -> Column
+ DataFrame.Internal.Column.Conversion: throwTypeMismatch :: (Typeable a, Typeable b) => Either DataFrameException Column
+ DataFrame.Internal.Column.Conversion: toDoubleVector :: Column -> Either DataFrameException (Vector Double)
+ DataFrame.Internal.Column.Conversion: toFloatVector :: Column -> Either DataFrameException (Vector Float)
+ DataFrame.Internal.Column.Conversion: toIntVector :: Column -> Either DataFrameException (Vector Int)
+ DataFrame.Internal.Column.Conversion: toList :: Columnable a => Column -> [a]
+ DataFrame.Internal.Column.Conversion: toUnboxedVector :: (Columnable a, Unbox a) => Column -> Either DataFrameException (Vector a)
+ DataFrame.Internal.Column.Conversion: toVector :: forall a v. (Vector v a, Columnable a) => Column -> Either DataFrameException (v a)
+ DataFrame.Internal.Column.Encode: dictCompactColumn :: Column -> Column
+ DataFrame.Internal.Column.Encode: dictEncodeColumn :: Column -> Maybe (Vector Int, Int)
+ DataFrame.Internal.Column.Encode: dictEncodeColumnUpTo :: Int -> Column -> Maybe (Vector Int, Int)
+ DataFrame.Internal.Column.Encode: dictMaxCardinality :: Int
+ DataFrame.Internal.Column.Merge: TextChunk :: !Array -> !Int -> !Vector Int -> !Maybe Bitmap -> TextChunk
+ DataFrame.Internal.Column.Merge: [tcBitmap] :: TextChunk -> !Maybe Bitmap
+ DataFrame.Internal.Column.Merge: [tcBytes] :: TextChunk -> !Array
+ DataFrame.Internal.Column.Merge: [tcOffsets] :: TextChunk -> !Vector Int
+ DataFrame.Internal.Column.Merge: [tcUsed] :: TextChunk -> !Int
+ DataFrame.Internal.Column.Merge: concatColumns :: [Column] -> Column
+ DataFrame.Internal.Column.Merge: concatValidity :: [Validity] -> Maybe Bitmap
+ DataFrame.Internal.Column.Merge: data TextChunk
+ DataFrame.Internal.Column.Merge: mergeTextChunks :: [TextChunk] -> Column
+ DataFrame.Internal.Column.Merge: packedFromTextChunk :: TextChunk -> Column
+ DataFrame.Internal.Column.Merge: tcRows :: TextChunk -> Int
+ DataFrame.Internal.Column.Operations: MGDouble :: !Vector Double -> MGSpec
+ DataFrame.Internal.Column.Operations: MGInt :: !Vector Int -> MGSpec
+ DataFrame.Internal.Column.Operations: MGSel32 :: !PackedTextData -> !Vector Int32 -> MGSpec
+ DataFrame.Internal.Column.Operations: MGSel64To32 :: !PackedTextData -> !Vector Int -> MGSpec
+ DataFrame.Internal.Column.Operations: MGSel64To64 :: !PackedTextData -> !Vector Int -> MGSpec
+ DataFrame.Internal.Column.Operations: allNullBoxed :: Int -> Vector a -> Vector a
+ DataFrame.Internal.Column.Operations: allNullUnboxed :: Unbox a => Int -> Vector a -> Vector a
+ DataFrame.Internal.Column.Operations: atIndicesStable :: Vector Int -> Column -> Column
+ DataFrame.Internal.Column.Operations: atIndicesStableMulti :: Vector Int -> [Column] -> [Column]
+ DataFrame.Internal.Column.Operations: checkMergedNoBothNull :: Column -> Column -> ()
+ DataFrame.Internal.Column.Operations: concatManyColumns :: [Column] -> Column
+ DataFrame.Internal.Column.Operations: copyIntoMutableColumn :: MutableColumn -> Int -> Column -> IO ()
+ DataFrame.Internal.Column.Operations: data MGSpec
+ DataFrame.Internal.Column.Operations: ensureOptional :: Column -> Column
+ DataFrame.Internal.Column.Operations: expandColumn :: Int -> Column -> Column
+ DataFrame.Internal.Column.Operations: fillGenerate :: Unbox c => IOVector c -> (Int -> c) -> Int -> Int -> IO ()
+ DataFrame.Internal.Column.Operations: findIndices :: Columnable a => (a -> Bool) -> Column -> Either DataFrameException (Vector Int)
+ DataFrame.Internal.Column.Operations: foldLinearGroups :: (Columnable b, Columnable acc) => (acc -> b -> acc) -> acc -> Column -> Vector Int -> Int -> Either DataFrameException Column
+ DataFrame.Internal.Column.Operations: foldl1Column :: Columnable a => (a -> a -> a) -> Column -> Either DataFrameException a
+ DataFrame.Internal.Column.Operations: foldl1DirectGroups :: Columnable a => (a -> a -> a) -> Column -> Vector Int -> Vector Int -> Either DataFrameException Column
+ DataFrame.Internal.Column.Operations: foldlColumn :: (Columnable a, Columnable b) => (b -> a -> b) -> b -> Column -> Either DataFrameException b
+ DataFrame.Internal.Column.Operations: forceColumn :: Column -> ()
+ DataFrame.Internal.Column.Operations: freezeColumnEither :: [(Int, Text)] -> MutableColumn -> IO Column
+ DataFrame.Internal.Column.Operations: freezeMutableColumn :: MutableColumn -> IO Column
+ DataFrame.Internal.Column.Operations: gatherWithSentinel :: Vector Int -> Column -> Column
+ DataFrame.Internal.Column.Operations: getIndices :: Vector Int -> Vector a -> Vector a
+ DataFrame.Internal.Column.Operations: getIndicesUnboxed :: Unbox a => Vector Int -> Vector a -> Vector a
+ DataFrame.Internal.Column.Operations: headColumn :: Columnable a => Column -> Either DataFrameException a
+ DataFrame.Internal.Column.Operations: ifoldrColumn :: (Columnable a, Columnable b) => (Int -> a -> b -> b) -> b -> Column -> Either DataFrameException b
+ DataFrame.Internal.Column.Operations: imapColumn :: (Columnable b, Columnable c) => (Int -> b -> c) -> Column -> Either DataFrameException Column
+ DataFrame.Internal.Column.Operations: leftExpandColumn :: Int -> Column -> Column
+ DataFrame.Internal.Column.Operations: mapColumn :: (Columnable b, Columnable c) => (b -> c) -> Column -> Either DataFrameException Column
+ DataFrame.Internal.Column.Operations: mappendColumns :: Column -> Column -> Either DataFrameException Column
+ DataFrame.Internal.Column.Operations: mappendColumnsEither :: Column -> Column -> Column
+ DataFrame.Internal.Column.Operations: mgClassify :: Column -> Maybe MGSpec
+ DataFrame.Internal.Column.Operations: mgFillDouble :: Vector Int -> Vector Double -> IOVector Double -> Int -> Int -> IO ()
+ DataFrame.Internal.Column.Operations: mgFillInt :: Vector Int -> Vector Int -> IOVector Int -> Int -> Int -> IO ()
+ DataFrame.Internal.Column.Operations: mgFillSel32 :: Vector Int -> Vector Int32 -> Int -> IOVector Int32 -> Int -> Int -> IO ()
+ DataFrame.Internal.Column.Operations: mgFillSel64To32 :: Vector Int -> Vector Int -> Int -> IOVector Int32 -> Int -> Int -> IO ()
+ DataFrame.Internal.Column.Operations: mgFillSel64To64 :: Vector Int -> Vector Int -> Int -> IOVector Int -> Int -> Int -> IO ()
+ DataFrame.Internal.Column.Operations: mgInt32Max :: Int
+ DataFrame.Internal.Column.Operations: mgOpen :: Int -> Vector Int -> MGSpec -> IO (Int -> Int -> IO (), IO Column)
+ DataFrame.Internal.Column.Operations: multiGatherRun :: Vector Int -> [MGSpec] -> Vector Column
+ DataFrame.Internal.Column.Operations: newMutableColumn :: Int -> Column -> IO MutableColumn
+ DataFrame.Internal.Column.Operations: parBackpermute2Int :: Vector Int -> Vector Int -> Vector Int
+ DataFrame.Internal.Column.Operations: parBackpermuteBoxed :: Vector a -> Vector Int -> Vector a
+ DataFrame.Internal.Column.Operations: parBackpermuteDouble :: Vector Double -> Vector Int -> Vector Double
+ DataFrame.Internal.Column.Operations: parBackpermuteInt :: Vector Int -> Vector Int -> Vector Int
+ DataFrame.Internal.Column.Operations: parBackpermuteUnboxed :: Unbox a => Vector a -> Vector Int -> Vector a
+ DataFrame.Internal.Column.Operations: parBitmapNonNeg :: Vector Int -> Bitmap
+ DataFrame.Internal.Column.Operations: parClampNonNeg :: Vector Int -> Vector Int
+ DataFrame.Internal.Column.Operations: parGenerateUnboxed :: Unbox c => Int -> (Int -> c) -> Vector c
+ DataFrame.Internal.Column.Operations: parGenerateUnboxedInline :: Unbox c => Int -> (Int -> c) -> Vector c
+ DataFrame.Internal.Column.Operations: sliceColumn :: Int -> Int -> Column -> Column
+ DataFrame.Internal.Column.Operations: takeLastColumn :: Int -> Column -> Column
+ DataFrame.Internal.Column.Operations: zipColumns :: Column -> Column -> Column
+ DataFrame.Internal.Column.Operations: zipWithColumns :: (Columnable a, Columnable b, Columnable c) => (a -> b -> c) -> Column -> Column -> Either DataFrameException Column
+ DataFrame.Internal.Column.Operations: zipWithColumnsGeneral :: (Columnable a, Columnable b, Columnable c) => (a -> b -> c) -> Column -> Column -> Either DataFrameException Column
+ DataFrame.Internal.Column.Properties: allMissing :: Column -> Bool
+ DataFrame.Internal.Column.Properties: columnElemIsNull :: Column -> Int -> Bool
+ DataFrame.Internal.Column.Properties: columnTypeString :: Column -> String
+ DataFrame.Internal.Column.Properties: columnVersionString :: Column -> String
+ DataFrame.Internal.Column.Properties: hasElemType :: Columnable a => Column -> Bool
+ DataFrame.Internal.Column.Properties: hasMissing :: Column -> Bool
+ DataFrame.Internal.Column.Properties: isMergedColumn :: Column -> Bool
+ DataFrame.Internal.Column.Properties: isNumeric :: Column -> Bool
+ DataFrame.Internal.Column.Properties: isPackedText :: Column -> Bool
+ DataFrame.Internal.Column.Properties: numElements :: Column -> Int
+ DataFrame.Internal.Column.Types: RBoxed :: Rep
+ DataFrame.Internal.Column.Types: RNullableBoxed :: Rep
+ DataFrame.Internal.Column.Types: RUnboxed :: Rep
+ DataFrame.Internal.Column.Types: That :: b -> These a b
+ DataFrame.Internal.Column.Types: These :: a -> b -> These a b
+ DataFrame.Internal.Column.Types: This :: a -> These a b
+ DataFrame.Internal.Column.Types: [SFalse] :: SBool 'False
+ DataFrame.Internal.Column.Types: [STrue] :: SBool 'True
+ DataFrame.Internal.Column.Types: class SBoolI (b :: Bool)
+ DataFrame.Internal.Column.Types: data Rep
+ DataFrame.Internal.Column.Types: data SBool (b :: Bool)
+ DataFrame.Internal.Column.Types: data These a b
+ DataFrame.Internal.Column.Types: instance (GHC.Classes.Eq a, GHC.Classes.Eq b) => GHC.Classes.Eq (DataFrame.Internal.Column.Types.These a b)
+ DataFrame.Internal.Column.Types: instance (GHC.Classes.Ord a, GHC.Classes.Ord b) => GHC.Classes.Ord (DataFrame.Internal.Column.Types.These a b)
+ DataFrame.Internal.Column.Types: instance (GHC.Read.Read a, GHC.Read.Read b) => GHC.Read.Read (DataFrame.Internal.Column.Types.These a b)
+ DataFrame.Internal.Column.Types: instance (GHC.Show.Show a, GHC.Show.Show b) => GHC.Show.Show (DataFrame.Internal.Column.Types.These a b)
+ DataFrame.Internal.Column.Types: instance Data.Foldable.Foldable (DataFrame.Internal.Column.Types.These a)
+ DataFrame.Internal.Column.Types: instance Data.Traversable.Traversable (DataFrame.Internal.Column.Types.These a)
+ DataFrame.Internal.Column.Types: instance DataFrame.Internal.Column.Types.SBoolI 'GHC.Types.False
+ DataFrame.Internal.Column.Types: instance DataFrame.Internal.Column.Types.SBoolI 'GHC.Types.True
+ DataFrame.Internal.Column.Types: instance GHC.Base.Functor (DataFrame.Internal.Column.Types.These a)
+ DataFrame.Internal.Column.Types: sFloating :: SBoolI (FloatingTypes a) => SBool (FloatingTypes a)
+ DataFrame.Internal.Column.Types: sIntegral :: SBoolI (IntegralTypes a) => SBool (IntegralTypes a)
+ DataFrame.Internal.Column.Types: sNumeric :: SBoolI (Numeric a) => SBool (Numeric a)
+ DataFrame.Internal.Column.Types: sUnbox :: SBoolI (Unboxable a) => SBool (Unboxable a)
+ DataFrame.Internal.Column.Types: sbool :: SBoolI b => SBool b
+ DataFrame.Internal.Column.Types: type Columnable' a = (Typeable a, Show a, Eq a)
+ DataFrame.Internal.Column.Types: type FloatingIf a = When FloatingTypes a (Real a, Fractional a)
+ DataFrame.Internal.Column.Types: type IntegralIf a = When IntegralTypes a Integral a
+ DataFrame.Internal.Column.Types: type UnboxIf a = When Unboxable a Unbox a
+ DataFrame.Internal.Column.Types: type family PromoteDiv a b
+ DataFrame.Internal.Control.Concurrent: boundsChunks :: Int -> Vector Int -> [(Int, Int)]
+ DataFrame.Internal.Control.Concurrent: capabilities :: Int
+ DataFrame.Internal.Control.Concurrent: capabilitiesIO :: IO Int
+ DataFrame.Internal.Control.Concurrent: chunksFor :: Int -> Int -> [(Int, Int)]
+ DataFrame.Internal.Control.Concurrent: forkJoin :: [IO a] -> IO [a]
+ DataFrame.Internal.Control.Concurrent: forkJoin_ :: [IO ()] -> IO ()
+ DataFrame.Internal.Control.Concurrent: parThreshold :: Int
+ DataFrame.Internal.Control.Concurrent: parallelBounds_ :: Int -> Vector Int -> (Int -> Int -> IO ()) -> IO ()
+ DataFrame.Internal.Control.Concurrent: parallelChunks :: Int -> Int -> (Int -> Int -> IO a) -> IO [a]
+ DataFrame.Internal.Control.Concurrent: parallelChunks_ :: Int -> Int -> (Int -> Int -> IO ()) -> IO ()
+ DataFrame.Internal.Control.Concurrent: pooledIndices :: Int -> Int -> (Int -> IO ()) -> IO ()
+ DataFrame.Internal.Control.Concurrent: pooledRun :: Int -> [IO a] -> IO [a]
+ DataFrame.Internal.Control.Concurrent: shouldParallelize :: Int -> Int -> Bool
+ DataFrame.Internal.Control.Concurrent: splitChunkRange :: Int -> Int -> [(Int, Int)]
+ DataFrame.Internal.Data.HashTable: HashTable :: !MVector s Int -> !MVector s Int -> !MVector s Int -> !Int -> HashTable s
+ DataFrame.Internal.Data.HashTable: [htGroup] :: HashTable s -> !MVector s Int
+ DataFrame.Internal.Data.HashTable: [htHash] :: HashTable s -> !MVector s Int
+ DataFrame.Internal.Data.HashTable: [htMask] :: HashTable s -> !Int
+ DataFrame.Internal.Data.HashTable: [htRep] :: HashTable s -> !MVector s Int
+ DataFrame.Internal.Data.HashTable: data HashTable s
+ DataFrame.Internal.Data.HashTable: htInsert :: PrimMonad m => HashTable (PrimState m) -> (Int -> Int -> Bool) -> Int -> Int -> Int -> m (Int, Bool)
+ DataFrame.Internal.Data.HashTable: newHashTable :: PrimMonad m => Int -> m (HashTable (PrimState m))
+ DataFrame.Internal.Data.HashTable: nextPow2Above :: Int -> Int
+ DataFrame.Internal.Data.PackedText: Offs32 :: {-# UNPACK #-} !Vector Int32 -> PackedOffsets
+ DataFrame.Internal.Data.PackedText: Offs64 :: {-# UNPACK #-} !Vector Int -> PackedOffsets
+ DataFrame.Internal.Data.PackedText: PackedTextData :: {-# UNPACK #-} !Array -> !PackedOffsets -> !Maybe PackedSel -> !Bool -> PackedTextData
+ DataFrame.Internal.Data.PackedText: Sel32 :: {-# UNPACK #-} !Vector Int32 -> PackedSel
+ DataFrame.Internal.Data.PackedText: Sel64 :: {-# UNPACK #-} !Vector Int -> PackedSel
+ DataFrame.Internal.Data.PackedText: [ptBytes] :: PackedTextData -> {-# UNPACK #-} !Array
+ DataFrame.Internal.Data.PackedText: [ptCanonicalSel] :: PackedTextData -> !Bool
+ DataFrame.Internal.Data.PackedText: [ptOffsets] :: PackedTextData -> !PackedOffsets
+ DataFrame.Internal.Data.PackedText: [ptSel] :: PackedTextData -> !Maybe PackedSel
+ DataFrame.Internal.Data.PackedText: data PackedOffsets
+ DataFrame.Internal.Data.PackedText: data PackedSel
+ DataFrame.Internal.Data.PackedText: data PackedTextData
+ DataFrame.Internal.Data.PackedText: mkOffsets :: Vector Int -> PackedOffsets
+ DataFrame.Internal.Data.PackedText: mkPackedContiguous :: Array -> Vector Int -> PackedTextData
+ DataFrame.Internal.Data.PackedText: mkPackedContiguous32 :: Array -> Vector Int32 -> PackedTextData
+ DataFrame.Internal.Data.PackedText: mkSel :: Int -> Vector Int -> PackedSel
+ DataFrame.Internal.Data.PackedText: offAt :: PackedOffsets -> Int -> Int
+ DataFrame.Internal.Data.PackedText: offCount :: PackedOffsets -> Int
+ DataFrame.Internal.Data.PackedText: packedGather :: Vector Int -> PackedTextData -> PackedTextData
+ DataFrame.Internal.Data.PackedText: packedIndexText :: PackedTextData -> Int -> Text
+ DataFrame.Internal.Data.PackedText: packedLength :: PackedTextData -> Int
+ DataFrame.Internal.Data.PackedText: packedRowOffsets :: PackedTextData -> Maybe (Array, PackedOffsets)
+ DataFrame.Internal.Data.PackedText: packedSlice :: PackedTextData -> Int -> (Array, Int, Int)
+ DataFrame.Internal.Data.PackedText: packedTake :: Int -> PackedTextData -> PackedTextData
+ DataFrame.Internal.Data.PackedText: selAt :: PackedSel -> Int -> Int
+ DataFrame.Internal.Data.PackedText: selLength :: PackedSel -> Int
+ DataFrame.Internal.Data.PackedText: sliceCmpBytes :: Array -> Int -> Int -> Array -> Int -> Int -> Ordering
+ DataFrame.Internal.Data.PackedText: sliceEqBytes :: Array -> Int -> Int -> Array -> Int -> Int -> Bool
+ DataFrame.Internal.Data.PackedText.Utf8: isUtf8Boundary :: Word8 -> Bool
+ DataFrame.Internal.Data.PackedText.Utf8: isValidUtf8Slice :: Array -> Int -> Int -> Bool
+ DataFrame.Internal.Data.PackedText.Utf8: lenientDecodeSlice :: Array -> Int -> Int -> Text
+ DataFrame.Internal.Data.PackedText.Utf8: sliceTextVector :: Array -> Vector Int -> Vector Text
+ DataFrame.Internal.DataFrame: GroupedInternal :: DataFrame -> [Text] -> Vector Int -> Vector Int -> Vector Int -> Vector Int -> GroupedDataFrame
+ DataFrame.Internal.DataFrame: [groupRepRows] :: GroupedDataFrame -> Vector Int
+ DataFrame.Internal.DataFrame: pattern Grouped :: DataFrame -> [Text] -> Vector Int -> Vector Int -> Vector Int -> GroupedDataFrame
+ DataFrame.Internal.Display.Pretty: (<+>) :: Doc -> Doc -> Doc
+ DataFrame.Internal.Display.Pretty: data Doc
+ DataFrame.Internal.Display.Pretty: defaultWidth :: Int
+ DataFrame.Internal.Display.Pretty: group :: Doc -> Doc
+ DataFrame.Internal.Display.Pretty: hardline :: Doc
+ DataFrame.Internal.Display.Pretty: hcat :: [Doc] -> Doc
+ DataFrame.Internal.Display.Pretty: infixr 6 <+>
+ DataFrame.Internal.Display.Pretty: instance GHC.Base.Monoid DataFrame.Internal.Display.Pretty.Doc
+ DataFrame.Internal.Display.Pretty: instance GHC.Base.Semigroup DataFrame.Internal.Display.Pretty.Doc
+ DataFrame.Internal.Display.Pretty: line :: Doc
+ DataFrame.Internal.Display.Pretty: nest :: Int -> Doc -> Doc
+ DataFrame.Internal.Display.Pretty: parens :: Doc -> Doc
+ DataFrame.Internal.Display.Pretty: parensWhenBroken :: Doc -> Doc
+ DataFrame.Internal.Display.Pretty: punctuate :: Doc -> [Doc] -> [Doc]
+ DataFrame.Internal.Display.Pretty: render :: Int -> Doc -> String
+ DataFrame.Internal.Display.Pretty: text :: String -> Doc
+ DataFrame.Internal.Expression: instance (Data.String.IsString a, DataFrame.Internal.Column.Base.Columnable a) => Data.String.IsString (DataFrame.Internal.Expression.Expr a)
+ DataFrame.Internal.Expression: instance (GHC.Float.Floating a, DataFrame.Internal.Column.Base.Columnable a) => GHC.Float.Floating (DataFrame.Internal.Expression.Expr a)
+ DataFrame.Internal.Expression: instance (GHC.Num.Num a, DataFrame.Internal.Column.Base.Columnable a) => GHC.Num.Num (DataFrame.Internal.Expression.Expr a)
+ DataFrame.Internal.Expression: instance (GHC.Real.Fractional a, DataFrame.Internal.Column.Base.Columnable a) => GHC.Real.Fractional (DataFrame.Internal.Expression.Expr a)
+ DataFrame.Internal.Expression.Operators: (.&&) :: (NullableCmpOp a b (NullCmpResult a b), BaseType a ~ Bool) => Expr a -> Expr b -> Expr (NullCmpResult a b)
+ DataFrame.Internal.Expression.Operators: (.&&.) :: Expr Bool -> Expr Bool -> Expr Bool
+ DataFrame.Internal.Expression.Operators: (.*) :: (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b (Promote (BaseType a) (BaseType b)) (WidenResult a b), Num (Promote (BaseType a) (BaseType b))) => Expr a -> Expr b -> Expr (WidenResult a b)
+ DataFrame.Internal.Expression.Operators: (.*.) :: (Columnable a, Num a) => Expr a -> Expr a -> Expr a
+ DataFrame.Internal.Expression.Operators: (.+) :: (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b (Promote (BaseType a) (BaseType b)) (WidenResult a b), Num (Promote (BaseType a) (BaseType b))) => Expr a -> Expr b -> Expr (WidenResult a b)
+ DataFrame.Internal.Expression.Operators: (.+.) :: (Columnable a, Num a) => Expr a -> Expr a -> Expr a
+ DataFrame.Internal.Expression.Operators: (.-) :: (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b (Promote (BaseType a) (BaseType b)) (WidenResult a b), Num (Promote (BaseType a) (BaseType b))) => Expr a -> Expr b -> Expr (WidenResult a b)
+ DataFrame.Internal.Expression.Operators: (.-.) :: (Columnable a, Num a) => Expr a -> Expr a -> Expr a
+ DataFrame.Internal.Expression.Operators: (./) :: (DivWidenOp (BaseType a) (BaseType b), NullLift2Op a b (PromoteDiv (BaseType a) (BaseType b)) (WidenResultDiv a b), Fractional (PromoteDiv (BaseType a) (BaseType b))) => Expr a -> Expr b -> Expr (WidenResultDiv a b)
+ DataFrame.Internal.Expression.Operators: (./.) :: (Columnable a, Fractional a) => Expr a -> Expr a -> Expr a
+ DataFrame.Internal.Expression.Operators: (./=) :: (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Eq (Promote (BaseType a) (BaseType b))) => Expr a -> Expr b -> Expr (NullCmpResult a b)
+ DataFrame.Internal.Expression.Operators: (./=.) :: (Columnable a, Eq a) => Expr a -> Expr a -> Expr Bool
+ DataFrame.Internal.Expression.Operators: (.<) :: (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Ord (Promote (BaseType a) (BaseType b))) => Expr a -> Expr b -> Expr (NullCmpResult a b)
+ DataFrame.Internal.Expression.Operators: (.<.) :: (Columnable a, Ord a) => Expr a -> Expr a -> Expr Bool
+ DataFrame.Internal.Expression.Operators: (.<=) :: (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Ord (Promote (BaseType a) (BaseType b))) => Expr a -> Expr b -> Expr (NullCmpResult a b)
+ DataFrame.Internal.Expression.Operators: (.<=.) :: (Columnable a, Ord a) => Expr a -> Expr a -> Expr Bool
+ DataFrame.Internal.Expression.Operators: (.=) :: Columnable a => Text -> Expr a -> NamedExpr
+ DataFrame.Internal.Expression.Operators: (.==) :: (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Eq (Promote (BaseType a) (BaseType b))) => Expr a -> Expr b -> Expr (NullCmpResult a b)
+ DataFrame.Internal.Expression.Operators: (.==.) :: (Columnable a, Eq a) => Expr a -> Expr a -> Expr Bool
+ DataFrame.Internal.Expression.Operators: (.>) :: (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Ord (Promote (BaseType a) (BaseType b))) => Expr a -> Expr b -> Expr (NullCmpResult a b)
+ DataFrame.Internal.Expression.Operators: (.>.) :: (Columnable a, Ord a) => Expr a -> Expr a -> Expr Bool
+ DataFrame.Internal.Expression.Operators: (.>=) :: (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Ord (Promote (BaseType a) (BaseType b))) => Expr a -> Expr b -> Expr (NullCmpResult a b)
+ DataFrame.Internal.Expression.Operators: (.>=.) :: (Columnable a, Ord a) => Expr a -> Expr a -> Expr Bool
+ DataFrame.Internal.Expression.Operators: (.^) :: (Columnable (BaseType a), Columnable (BaseType b), Num (BaseType a), Integral (BaseType b), NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b (BaseType a) a, Num (Promote (BaseType a) (BaseType b))) => Expr a -> Expr b -> Expr a
+ DataFrame.Internal.Expression.Operators: (.^.) :: (Columnable a, Columnable b, Num a, Integral b) => Expr a -> Expr b -> Expr a
+ DataFrame.Internal.Expression.Operators: (.^^) :: (Columnable (BaseType a), Columnable (BaseType b), Fractional (BaseType a), Integral (BaseType b), NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b (BaseType a) a, Num (Promote (BaseType a) (BaseType b))) => Expr a -> Expr b -> Expr a
+ DataFrame.Internal.Expression.Operators: (.^^.) :: (Columnable a, Columnable b, Fractional a, Integral b) => Expr a -> Expr b -> Expr a
+ DataFrame.Internal.Expression.Operators: (.||) :: (NullableCmpOp a b (NullCmpResult a b), BaseType a ~ Bool) => Expr a -> Expr b -> Expr (NullCmpResult a b)
+ DataFrame.Internal.Expression.Operators: (.||.) :: Expr Bool -> Expr Bool -> Expr Bool
+ DataFrame.Internal.Expression.Operators: (|>) :: a -> (a -> b) -> b
+ DataFrame.Internal.Expression.Operators: [NullAnd] :: forall a b. (NullableCmpOp a b (NullCmpResult a b), BaseType a ~ Bool) => NullAnd a b (NullCmpResult a b)
+ DataFrame.Internal.Expression.Operators: [NullEq] :: forall a b. (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Eq (Promote (BaseType a) (BaseType b))) => NullEq a b (NullCmpResult a b)
+ DataFrame.Internal.Expression.Operators: [NullGeq] :: forall a b. (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Ord (Promote (BaseType a) (BaseType b))) => NullGeq a b (NullCmpResult a b)
+ DataFrame.Internal.Expression.Operators: [NullGt] :: forall a b. (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Ord (Promote (BaseType a) (BaseType b))) => NullGt a b (NullCmpResult a b)
+ DataFrame.Internal.Expression.Operators: [NullLeq] :: forall a b. (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Ord (Promote (BaseType a) (BaseType b))) => NullLeq a b (NullCmpResult a b)
+ DataFrame.Internal.Expression.Operators: [NullLt] :: forall a b. (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Ord (Promote (BaseType a) (BaseType b))) => NullLt a b (NullCmpResult a b)
+ DataFrame.Internal.Expression.Operators: [NullNeq] :: forall a b. (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Eq (Promote (BaseType a) (BaseType b))) => NullNeq a b (NullCmpResult a b)
+ DataFrame.Internal.Expression.Operators: [NullOr] :: forall a b. (NullableCmpOp a b (NullCmpResult a b), BaseType a ~ Bool) => NullOr a b (NullCmpResult a b)
+ DataFrame.Internal.Expression.Operators: as :: Columnable a => Expr a -> Text -> NamedExpr
+ DataFrame.Internal.Expression.Operators: col :: Columnable a => Text -> Expr a
+ DataFrame.Internal.Expression.Operators: data NullAnd a b c
+ DataFrame.Internal.Expression.Operators: data NullEq a b c
+ DataFrame.Internal.Expression.Operators: data NullGeq a b c
+ DataFrame.Internal.Expression.Operators: data NullGt a b c
+ DataFrame.Internal.Expression.Operators: data NullLeq a b c
+ DataFrame.Internal.Expression.Operators: data NullLt a b c
+ DataFrame.Internal.Expression.Operators: data NullNeq a b c
+ DataFrame.Internal.Expression.Operators: data NullOr a b c
+ DataFrame.Internal.Expression.Operators: ifThenElse :: Columnable a => Expr Bool -> Expr a -> Expr a -> Expr a
+ DataFrame.Internal.Expression.Operators: infix 4 .>=
+ DataFrame.Internal.Expression.Operators: infixl 6 .-
+ DataFrame.Internal.Expression.Operators: infixl 7 ./
+ DataFrame.Internal.Expression.Operators: infixr 0 .=
+ DataFrame.Internal.Expression.Operators: infixr 2 .||
+ DataFrame.Internal.Expression.Operators: infixr 3 .&&
+ DataFrame.Internal.Expression.Operators: infixr 8 .^.
+ DataFrame.Internal.Expression.Operators: instance DataFrame.Internal.Expression.BinaryOp DataFrame.Internal.Expression.Operators.NullAnd
+ DataFrame.Internal.Expression.Operators: instance DataFrame.Internal.Expression.BinaryOp DataFrame.Internal.Expression.Operators.NullEq
+ DataFrame.Internal.Expression.Operators: instance DataFrame.Internal.Expression.BinaryOp DataFrame.Internal.Expression.Operators.NullGeq
+ DataFrame.Internal.Expression.Operators: instance DataFrame.Internal.Expression.BinaryOp DataFrame.Internal.Expression.Operators.NullGt
+ DataFrame.Internal.Expression.Operators: instance DataFrame.Internal.Expression.BinaryOp DataFrame.Internal.Expression.Operators.NullLeq
+ DataFrame.Internal.Expression.Operators: instance DataFrame.Internal.Expression.BinaryOp DataFrame.Internal.Expression.Operators.NullLt
+ DataFrame.Internal.Expression.Operators: instance DataFrame.Internal.Expression.BinaryOp DataFrame.Internal.Expression.Operators.NullNeq
+ DataFrame.Internal.Expression.Operators: instance DataFrame.Internal.Expression.BinaryOp DataFrame.Internal.Expression.Operators.NullOr
+ DataFrame.Internal.Expression.Operators: lift2Decorated :: (Columnable c, Columnable b, Columnable a) => (c -> b -> a) -> Text -> Maybe Text -> Bool -> Int -> Expr c -> Expr b -> Expr a
+ DataFrame.Internal.Expression.Operators: liftDecorated :: (Columnable a, Columnable b) => (a -> b) -> Text -> Maybe Text -> Expr a -> Expr b
+ DataFrame.Internal.Expression.Operators: lit :: Columnable a => a -> Expr a
+ DataFrame.Internal.Expression.Operators: name :: Show a => Expr a -> Text
+ DataFrame.Internal.Expression.Operators.Nullable: applyNull1 :: NullLift1Op a r c => (BaseType a -> r) -> a -> c
+ DataFrame.Internal.Expression.Operators.Nullable: applyNull2 :: NullLift2Op a b r c => (BaseType a -> BaseType b -> r) -> a -> b -> c
+ DataFrame.Internal.Expression.Operators.Nullable: class Columnable PromoteDiv a b => DivWidenOp a b
+ DataFrame.Internal.Expression.Operators.Nullable: class (Columnable a, Columnable r, Columnable c) => NullLift1Op a r c
+ DataFrame.Internal.Expression.Operators.Nullable: class (Columnable a, Columnable b, Columnable r, Columnable c) => NullLift2Op a b r c
+ DataFrame.Internal.Expression.Operators.Nullable: class (Columnable a, Columnable b, Columnable c) => NullableArithOp a b c | a b -> c
+ DataFrame.Internal.Expression.Operators.Nullable: class (Columnable a, Columnable b, Columnable e) => NullableCmpOp a b e
+ DataFrame.Internal.Expression.Operators.Nullable: class Columnable Promote a b => NumericWidenOp a b
+ DataFrame.Internal.Expression.Operators.Nullable: divArithOp :: DivWidenOp a b => (PromoteDiv a b -> PromoteDiv a b -> PromoteDiv a b) -> a -> b -> PromoteDiv a b
+ DataFrame.Internal.Expression.Operators.Nullable: divWiden1 :: DivWidenOp a b => a -> PromoteDiv a b
+ DataFrame.Internal.Expression.Operators.Nullable: divWiden2 :: DivWidenOp a b => b -> PromoteDiv a b
+ DataFrame.Internal.Expression.Operators.Nullable: instance (DataFrame.Internal.Column.Base.Columnable a, DataFrame.Internal.Column.Base.Columnable (GHC.Maybe.Maybe a)) => DataFrame.Internal.Expression.Operators.Nullable.NullableArithOp (GHC.Maybe.Maybe a) (GHC.Maybe.Maybe a) (GHC.Maybe.Maybe a)
+ DataFrame.Internal.Expression.Operators.Nullable: instance (DataFrame.Internal.Column.Base.Columnable a, DataFrame.Internal.Column.Base.Columnable (GHC.Maybe.Maybe a)) => DataFrame.Internal.Expression.Operators.Nullable.NullableArithOp (GHC.Maybe.Maybe a) a (GHC.Maybe.Maybe a)
+ DataFrame.Internal.Expression.Operators.Nullable: instance (DataFrame.Internal.Column.Base.Columnable a, DataFrame.Internal.Column.Base.Columnable (GHC.Maybe.Maybe a), DataFrame.Internal.Column.Base.Columnable (GHC.Maybe.Maybe GHC.Types.Bool)) => DataFrame.Internal.Expression.Operators.Nullable.NullableCmpOp (GHC.Maybe.Maybe a) (GHC.Maybe.Maybe a) (GHC.Maybe.Maybe GHC.Types.Bool)
+ DataFrame.Internal.Expression.Operators.Nullable: instance (DataFrame.Internal.Column.Base.Columnable a, DataFrame.Internal.Column.Base.Columnable (GHC.Maybe.Maybe a), DataFrame.Internal.Column.Base.Columnable (GHC.Maybe.Maybe GHC.Types.Bool)) => DataFrame.Internal.Expression.Operators.Nullable.NullableCmpOp (GHC.Maybe.Maybe a) a (GHC.Maybe.Maybe GHC.Types.Bool)
+ DataFrame.Internal.Expression.Operators.Nullable: instance (DataFrame.Internal.Column.Base.Columnable a, DataFrame.Internal.Column.Base.Columnable (GHC.Maybe.Maybe a), DataFrame.Internal.Column.Base.Columnable (GHC.Maybe.Maybe GHC.Types.Bool), a GHC.Types.~ DataFrame.Internal.Expression.Operators.Nullable.BaseType a) => DataFrame.Internal.Expression.Operators.Nullable.NullableCmpOp a (GHC.Maybe.Maybe a) (GHC.Maybe.Maybe GHC.Types.Bool)
+ DataFrame.Internal.Expression.Operators.Nullable: instance (DataFrame.Internal.Column.Base.Columnable a, DataFrame.Internal.Column.Base.Columnable (GHC.Maybe.Maybe a), a GHC.Types.~ DataFrame.Internal.Expression.Operators.Nullable.BaseType a) => DataFrame.Internal.Expression.Operators.Nullable.NullableArithOp a (GHC.Maybe.Maybe a) (GHC.Maybe.Maybe a)
+ DataFrame.Internal.Expression.Operators.Nullable: instance (DataFrame.Internal.Column.Base.Columnable a, DataFrame.Internal.Column.Base.Columnable GHC.Types.Bool, a GHC.Types.~ DataFrame.Internal.Expression.Operators.Nullable.BaseType a) => DataFrame.Internal.Expression.Operators.Nullable.NullableCmpOp a a GHC.Types.Bool
+ DataFrame.Internal.Expression.Operators.Nullable: instance (DataFrame.Internal.Column.Base.Columnable a, DataFrame.Internal.Column.Base.Columnable b, DataFrame.Internal.Column.Base.Columnable r, DataFrame.Internal.Column.Base.Columnable (GHC.Maybe.Maybe r)) => DataFrame.Internal.Expression.Operators.Nullable.NullLift2Op (GHC.Maybe.Maybe a) (GHC.Maybe.Maybe b) r (GHC.Maybe.Maybe r)
+ DataFrame.Internal.Expression.Operators.Nullable: instance (DataFrame.Internal.Column.Base.Columnable a, DataFrame.Internal.Column.Base.Columnable b, DataFrame.Internal.Column.Base.Columnable r, DataFrame.Internal.Column.Base.Columnable (GHC.Maybe.Maybe r), a GHC.Types.~ DataFrame.Internal.Expression.Operators.Nullable.BaseType a) => DataFrame.Internal.Expression.Operators.Nullable.NullLift2Op a (GHC.Maybe.Maybe b) r (GHC.Maybe.Maybe r)
+ DataFrame.Internal.Expression.Operators.Nullable: instance (DataFrame.Internal.Column.Base.Columnable a, DataFrame.Internal.Column.Base.Columnable b, DataFrame.Internal.Column.Base.Columnable r, DataFrame.Internal.Column.Base.Columnable (GHC.Maybe.Maybe r), b GHC.Types.~ DataFrame.Internal.Expression.Operators.Nullable.BaseType b) => DataFrame.Internal.Expression.Operators.Nullable.NullLift2Op (GHC.Maybe.Maybe a) b r (GHC.Maybe.Maybe r)
+ DataFrame.Internal.Expression.Operators.Nullable: instance (DataFrame.Internal.Column.Base.Columnable a, DataFrame.Internal.Column.Base.Columnable b, DataFrame.Internal.Column.Base.Columnable r, a GHC.Types.~ DataFrame.Internal.Expression.Operators.Nullable.BaseType a, b GHC.Types.~ DataFrame.Internal.Expression.Operators.Nullable.BaseType b) => DataFrame.Internal.Expression.Operators.Nullable.NullLift2Op a b r r
+ DataFrame.Internal.Expression.Operators.Nullable: instance (DataFrame.Internal.Column.Base.Columnable a, DataFrame.Internal.Column.Base.Columnable r, DataFrame.Internal.Column.Base.Columnable (GHC.Maybe.Maybe r)) => DataFrame.Internal.Expression.Operators.Nullable.NullLift1Op (GHC.Maybe.Maybe a) r (GHC.Maybe.Maybe r)
+ DataFrame.Internal.Expression.Operators.Nullable: instance (DataFrame.Internal.Column.Base.Columnable a, DataFrame.Internal.Column.Base.Columnable r, a GHC.Types.~ DataFrame.Internal.Expression.Operators.Nullable.BaseType a) => DataFrame.Internal.Expression.Operators.Nullable.NullLift1Op a r r
+ DataFrame.Internal.Expression.Operators.Nullable: instance (DataFrame.Internal.Column.Base.Columnable a, a GHC.Types.~ DataFrame.Internal.Expression.Operators.Nullable.BaseType a) => DataFrame.Internal.Expression.Operators.Nullable.NullableArithOp a a a
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Column.Base.Columnable a => DataFrame.Internal.Expression.Operators.Nullable.NumericWidenOp a a
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.DivWidenOp GHC.Int.Int32 GHC.Int.Int32
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.DivWidenOp GHC.Int.Int32 GHC.Int.Int64
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.DivWidenOp GHC.Int.Int32 GHC.Types.Double
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.DivWidenOp GHC.Int.Int32 GHC.Types.Float
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.DivWidenOp GHC.Int.Int32 GHC.Types.Int
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.DivWidenOp GHC.Int.Int64 GHC.Int.Int32
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.DivWidenOp GHC.Int.Int64 GHC.Int.Int64
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.DivWidenOp GHC.Int.Int64 GHC.Types.Double
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.DivWidenOp GHC.Int.Int64 GHC.Types.Float
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.DivWidenOp GHC.Int.Int64 GHC.Types.Int
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.DivWidenOp GHC.Types.Double GHC.Int.Int32
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.DivWidenOp GHC.Types.Double GHC.Int.Int64
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.DivWidenOp GHC.Types.Double GHC.Types.Double
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.DivWidenOp GHC.Types.Double GHC.Types.Float
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.DivWidenOp GHC.Types.Double GHC.Types.Int
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.DivWidenOp GHC.Types.Float GHC.Int.Int32
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.DivWidenOp GHC.Types.Float GHC.Int.Int64
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.DivWidenOp GHC.Types.Float GHC.Types.Double
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.DivWidenOp GHC.Types.Float GHC.Types.Float
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.DivWidenOp GHC.Types.Float GHC.Types.Int
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.DivWidenOp GHC.Types.Int GHC.Int.Int32
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.DivWidenOp GHC.Types.Int GHC.Int.Int64
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.DivWidenOp GHC.Types.Int GHC.Types.Double
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.DivWidenOp GHC.Types.Int GHC.Types.Float
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.DivWidenOp GHC.Types.Int GHC.Types.Int
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.NumericWidenOp GHC.Int.Int32 GHC.Types.Double
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.NumericWidenOp GHC.Int.Int32 GHC.Types.Float
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.NumericWidenOp GHC.Int.Int64 GHC.Types.Double
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.NumericWidenOp GHC.Int.Int64 GHC.Types.Float
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.NumericWidenOp GHC.Types.Double GHC.Int.Int32
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.NumericWidenOp GHC.Types.Double GHC.Int.Int64
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.NumericWidenOp GHC.Types.Double GHC.Types.Float
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.NumericWidenOp GHC.Types.Double GHC.Types.Int
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.NumericWidenOp GHC.Types.Float GHC.Int.Int32
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.NumericWidenOp GHC.Types.Float GHC.Int.Int64
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.NumericWidenOp GHC.Types.Float GHC.Types.Double
+ DataFrame.Internal.Expression.Operators.Nullable: instance DataFrame.Internal.Expression.Operators.Nullable.NumericWidenOp GHC.Types.Int GHC.Types.Double
+ DataFrame.Internal.Expression.Operators.Nullable: nullArithOp :: NullableArithOp a b c => (BaseType a -> BaseType a -> BaseType a) -> a -> b -> c
+ DataFrame.Internal.Expression.Operators.Nullable: nullCmpOp :: NullableCmpOp a b e => (BaseType a -> BaseType a -> Bool) -> a -> b -> e
+ DataFrame.Internal.Expression.Operators.Nullable: type WidenResult a b = NullLift2Result a b Promote BaseType a BaseType b
+ DataFrame.Internal.Expression.Operators.Nullable: type WidenResultDiv a b = NullLift2Result a b PromoteDiv BaseType a BaseType b
+ DataFrame.Internal.Expression.Operators.Nullable: type family NullCmpResult a b
+ DataFrame.Internal.Expression.Operators.Nullable: widen1 :: NumericWidenOp a b => a -> Promote a b
+ DataFrame.Internal.Expression.Operators.Nullable: widen2 :: NumericWidenOp a b => b -> Promote a b
+ DataFrame.Internal.Expression.Operators.Nullable: widenArithOp :: NumericWidenOp a b => (Promote a b -> Promote a b -> Promote a b) -> a -> b -> Promote a b
+ DataFrame.Internal.Expression.Operators.Nullable: widenCmpOp :: NumericWidenOp a b => (Promote a b -> Promote a b -> Bool) -> a -> b -> Bool
+ DataFrame.Internal.Expression.Simplify: data PredFact
+ DataFrame.Internal.Expression.Simplify: entails :: [PredFact] -> Expr Bool -> Maybe Bool
+ DataFrame.Internal.Expression.Simplify: factFalse :: Expr Bool -> Maybe PredFact
+ DataFrame.Internal.Expression.Simplify: factTrue :: Expr Bool -> Maybe PredFact
+ DataFrame.Internal.Expression.Simplify: instance GHC.Classes.Eq DataFrame.Internal.Expression.Simplify.Cmp
+ DataFrame.Internal.Expression.Simplify: instance GHC.Classes.Eq DataFrame.Internal.Expression.Simplify.NullK
+ DataFrame.Internal.Expression.Simplify: simplify :: Columnable a => Expr a -> Expr a
+ DataFrame.Internal.Expression.Simplify: simplifyPredicatePair :: Columnable a => Bool -> Expr a -> Expr a -> Maybe (Expr a)
+ DataFrame.Internal.Grouping.Direct: DirectGrouping :: !Vector Int -> !Vector Int -> !Vector Int -> !Int -> DirectGrouping
+ DataFrame.Internal.Grouping.Direct: [dgNGroups] :: DirectGrouping -> !Int
+ DataFrame.Internal.Grouping.Direct: [dgOffsets] :: DirectGrouping -> !Vector Int
+ DataFrame.Internal.Grouping.Direct: [dgRowToGroup] :: DirectGrouping -> !Vector Int
+ DataFrame.Internal.Grouping.Direct: [dgValueIndices] :: DirectGrouping -> !Vector Int
+ DataFrame.Internal.Grouping.Direct: ascendingCodeGroups :: Vector Int -> (Vector Int, Int)
+ DataFrame.Internal.Grouping.Direct: data DirectGrouping
+ DataFrame.Internal.Grouping.Direct: directGroupThreshold :: Int
+ DataFrame.Internal.Grouping.Direct: directLayoutLazy :: (Int -> Int) -> Int -> Int -> (Vector Int -> (Vector Int, Int)) -> Maybe (Vector Int, Vector Int, Vector Int, Int)
+ DataFrame.Internal.Grouping.Direct: groupCodesMaybe :: (Int -> Int) -> Int -> Int -> (Vector Int -> (Vector Int, Int)) -> Maybe DirectGrouping
+ DataFrame.Internal.Grouping.Direct: rangeOf :: Vector Int -> (Int, Int)
+ DataFrame.Internal.Grouping.Direct: tryDirectGroupColumn :: Column -> Maybe DirectGrouping
+ DataFrame.Internal.Grouping.Direct: useTwoLevel :: Int -> Int -> Bool
+ DataFrame.Internal.Grouping.Direct: visFromRowToGroup :: Int -> Int -> Vector Int -> Vector Int -> Vector Int
+ DataFrame.Internal.Grouping.Partitioned: numPartitionsFor :: Int -> Int -> Int
+ DataFrame.Internal.Grouping.Partitioned: parallelAssignGroups :: Int -> Vector Int -> (Int -> Int -> Bool) -> IO (Vector Int, Vector Int)
+ DataFrame.Internal.Grouping.Partitioned: rtgFromVisOffs :: Int -> Vector Int -> Vector Int -> Vector Int
+ DataFrame.Internal.Row.RowHash: computeRowHashesIO :: Int -> [Column] -> IO (Vector Int)
+ DataFrame.Internal.Row.RowHash: computeRowHashesWithIO :: Bool -> Int -> [Column] -> IO (Vector Int)
+ DataFrame.Internal.Row.RowHash: hashRowRange :: Bool -> IOVector Int -> Int -> Int -> [Column] -> IO ()
+ DataFrame.Internal.Row.RowHash: parRowHashThreshold :: Int
+ DataFrame.Typed.Generic: instance (GHC.TypeLits.KnownSymbol name, DataFrame.Internal.Column.Base.Columnable a) => DataFrame.Typed.Generic.GHasColumns (GHC.Generics.M1 GHC.Generics.S ('GHC.Generics.MetaSel ('GHC.Maybe.Just name) su ss ds) (GHC.Generics.K1 i a))
+ DataFrame.Typed.Schema: instance (GHC.TypeLits.KnownSymbol name, Data.Typeable.Internal.Typeable a, DataFrame.Internal.Column.Base.Columnable a, DataFrame.Typed.Schema.KnownSchema rest) => DataFrame.Typed.Schema.KnownSchema ('(name, a) : rest)
Files
- dataframe-core.cabal +38/−23
- src-internal/DataFrame/Errors.hs +2/−0
- src-internal/DataFrame/Internal/AggKernel.hs +0/−260
- src-internal/DataFrame/Internal/AggKernelDirect.hs +0/−338
- src-internal/DataFrame/Internal/AggKernelPar.hs +0/−391
- src-internal/DataFrame/Internal/AggPlan.hs +0/−299
- src-internal/DataFrame/Internal/Aggregation/Kernel/Dense.hs +832/−0
- src-internal/DataFrame/Internal/Aggregation/Kernel/Fused.hs +757/−0
- src-internal/DataFrame/Internal/Aggregation/Kernel/Moments.hs +389/−0
- src-internal/DataFrame/Internal/Aggregation/Kernel/Scatter.hs +818/−0
- src-internal/DataFrame/Internal/Aggregation/Plan.hs +222/−0
- src-internal/DataFrame/Internal/Aggregation/Reduction.hs +44/−0
- src-internal/DataFrame/Internal/Algorithms/Hash.hs +113/−0
- src-internal/DataFrame/Internal/Algorithms/Rank/Radix.hs +101/−0
- src-internal/DataFrame/Internal/Algorithms/Sort/Radix/Parallel.hs +332/−0
- src-internal/DataFrame/Internal/Column.hs +26/−1835
- src-internal/DataFrame/Internal/Column/Base.hs +333/−0
- src-internal/DataFrame/Internal/Column/Bitmap.hs +188/−0
- src-internal/DataFrame/Internal/Column/Builder.hs +306/−0
- src-internal/DataFrame/Internal/Column/Conversion.hs +441/−0
- src-internal/DataFrame/Internal/Column/Encode.hs +185/−0
- src-internal/DataFrame/Internal/Column/Merge.hs +138/−0
- src-internal/DataFrame/Internal/Column/Operations.hs +1369/−0
- src-internal/DataFrame/Internal/Column/Properties.hs +153/−0
- src-internal/DataFrame/Internal/Column/Types.hs +161/−0
- src-internal/DataFrame/Internal/ColumnBuilder.hs +0/−299
- src-internal/DataFrame/Internal/ColumnMerge.hs +0/−185
- src-internal/DataFrame/Internal/Control/Concurrent.hs +177/−0
- src-internal/DataFrame/Internal/Data/HashTable.hs +108/−0
- src-internal/DataFrame/Internal/Data/PackedText.hs +361/−0
- src-internal/DataFrame/Internal/Data/PackedText/Utf8.hs +95/−0
- src-internal/DataFrame/Internal/DataFrame.hs +100/−46
- src-internal/DataFrame/Internal/DictEncode.hs +0/−171
- src-internal/DataFrame/Internal/Display/Pretty.hs +129/−0
- src-internal/DataFrame/Internal/Expression.hs +1/−1
- src-internal/DataFrame/Internal/Expression/Operators.hs +425/−0
- src-internal/DataFrame/Internal/Expression/Operators/Nullable.hs +467/−0
- src-internal/DataFrame/Internal/Expression/Simplify.hs +417/−0
- src-internal/DataFrame/Internal/Grouping.hs +667/−78
- src-internal/DataFrame/Internal/Grouping/Direct.hs +859/−0
- src-internal/DataFrame/Internal/Grouping/Partitioned.hs +406/−0
- src-internal/DataFrame/Internal/GroupingDirect.hs +0/−234
- src-internal/DataFrame/Internal/GroupingPar.hs +0/−308
- src-internal/DataFrame/Internal/Hash.hs +0/−113
- src-internal/DataFrame/Internal/HashTable.hs +0/−94
- src-internal/DataFrame/Internal/Interpreter.hs +47/−5
- src-internal/DataFrame/Internal/Nullable.hs +0/−467
- src-internal/DataFrame/Internal/PackedText.hs +0/−238
- src-internal/DataFrame/Internal/ParRadixSort.hs +0/−272
- src-internal/DataFrame/Internal/Pretty.hs +0/−129
- src-internal/DataFrame/Internal/RadixRank.hs +0/−101
- src-internal/DataFrame/Internal/Row.hs +17/−5
- src-internal/DataFrame/Internal/Row/RowHash.hs +240/−0
- src-internal/DataFrame/Internal/RowHash.hs +0/−223
- src-internal/DataFrame/Internal/Simplify.hs +0/−417
- src-internal/DataFrame/Internal/Types.hs +0/−161
- src-internal/DataFrame/Internal/Utf8.hs +0/−95
- src-internal/DataFrame/Operators.hs +0/−425
- src/DataFrame/Core.hs +1/−1
- src/DataFrame/Expression/Operators.hs +12/−0
- src/DataFrame/Typed/Schema.hs +1/−1
- src/DataFrame/Typed/Types.hs +1/−1
dataframe-core.cabal view
@@ -1,6 +1,6 @@ cabal-version: 3.4 name: dataframe-core-version: 2.4.0.0+version: 2.5.0.0 synopsis: Core data structures for the dataframe library. description: Minimal interchange-format types for the @dataframe@ ecosystem:@@ -32,42 +32,57 @@ import: warnings exposed-modules: DataFrame.Core+ DataFrame.Errors+ DataFrame.Expression.Operators DataFrame.Typed.Freeze DataFrame.Typed.Generic DataFrame.Typed.Record DataFrame.Typed.Schema DataFrame.Typed.Types DataFrame.Typed.Util- DataFrame.Errors- DataFrame.Operators DataFrame.Display.Terminal.Colours DataFrame.Display.Terminal.PrettyPrint- DataFrame.Internal.AggKernel- DataFrame.Internal.AggKernelDirect- DataFrame.Internal.AggKernelPar- DataFrame.Internal.AggPlan- DataFrame.Internal.GroupingDirect+ -- Concurrency+ DataFrame.Internal.Control.Concurrent+ -- Aggregation kernels and planning.+ DataFrame.Internal.Aggregation.Kernel.Dense+ DataFrame.Internal.Aggregation.Kernel.Fused+ DataFrame.Internal.Aggregation.Kernel.Moments+ DataFrame.Internal.Aggregation.Kernel.Scatter+ DataFrame.Internal.Aggregation.Plan+ DataFrame.Internal.Aggregation.Reduction+ -- Shared algorithms.+ DataFrame.Internal.Algorithms.Hash+ DataFrame.Internal.Algorithms.Rank.Radix+ DataFrame.Internal.Algorithms.Sort.Radix.Parallel+ -- The column representation. DataFrame.Internal.Column- DataFrame.Internal.ColumnBuilder- DataFrame.Internal.ColumnMerge+ DataFrame.Internal.Column.Base+ DataFrame.Internal.Column.Bitmap+ DataFrame.Internal.Column.Builder+ DataFrame.Internal.Column.Conversion+ DataFrame.Internal.Column.Encode+ DataFrame.Internal.Column.Merge+ DataFrame.Internal.Column.Operations+ DataFrame.Internal.Column.Properties+ DataFrame.Internal.Column.Types+ -- Backing data structures.+ DataFrame.Internal.Data.HashTable+ DataFrame.Internal.Data.PackedText+ DataFrame.Internal.Data.PackedText.Utf8 DataFrame.Internal.DataFrame- DataFrame.Internal.DictEncode+ DataFrame.Internal.Display.Pretty+ -- Expressions: syntax, operators, simplification. DataFrame.Internal.Expression+ DataFrame.Internal.Expression.Operators+ DataFrame.Internal.Expression.Operators.Nullable+ DataFrame.Internal.Expression.Simplify DataFrame.Internal.Grouping- DataFrame.Internal.GroupingPar- DataFrame.Internal.Hash- DataFrame.Internal.HashTable+ DataFrame.Internal.Grouping.Direct+ DataFrame.Internal.Grouping.Partitioned DataFrame.Internal.Interpreter- DataFrame.Internal.Nullable- DataFrame.Internal.PackedText- DataFrame.Internal.ParRadixSort- DataFrame.Internal.Pretty- DataFrame.Internal.RadixRank- DataFrame.Internal.RowHash DataFrame.Internal.Row- DataFrame.Internal.Simplify- DataFrame.Internal.Types- DataFrame.Internal.Utf8+ DataFrame.Internal.Row.RowHash build-depends: base >= 4 && < 5, containers >= 0.6.7 && < 0.10, primitive >= 0.7 && < 0.11,
src-internal/DataFrame/Errors.hs view
@@ -34,6 +34,7 @@ ColumnsNotFoundException :: [T.Text] -> T.Text -> [T.Text] -> DataFrameException EmptyDataSetException :: T.Text -> DataFrameException InternalException :: T.Text -> DataFrameException+ ExpectedNonNullableException :: DataFrameException NonColumnReferenceException :: T.Text -> DataFrameException UnaggregatedException :: T.Text -> DataFrameException WrongQuantileNumberException :: Int -> DataFrameException@@ -42,6 +43,7 @@ instance Show DataFrameException where show :: DataFrameException -> String+ show ExpectedNonNullableException = "Expected non-nullable column" show (TypeMismatchException context) = let errorString =
− src-internal/DataFrame/Internal/AggKernel.hs
@@ -1,260 +0,0 @@-{-# LANGUAGE BangPatterns #-}-{-# LANGUAGE ExplicitNamespaces #-}-{-# LANGUAGE FlexibleContexts #-}-{-# LANGUAGE GADTs #-}-{-# LANGUAGE LambdaCase #-}-{-# LANGUAGE ScopedTypeVariables #-}-{-# LANGUAGE TypeApplications #-}---- | Vectorized scatter-accumulate aggregation kernel.-module DataFrame.Internal.AggKernel (- Reduction (..),- scatterReduce,- scatterColumnToDouble,-) where--import Data.Type.Equality (TestEquality (..), type (:~:) (Refl))-import qualified Data.Vector.Unboxed as VU-import qualified Data.Vector.Unboxed.Mutable as VUM--import Control.Monad (when)-import Control.Monad.ST (ST, runST)-import DataFrame.Internal.Column (- Column (..),- Columnable,- fromUnboxedVector,- materializePacked,- )-import Type.Reflection (typeRep)--{- | A recognised fast-path reduction over a single value column. The element-type (Int vs Double) is resolved at scatter time; sum/min/max preserve the-column's element type, everything else produces a Double column.--}-data Reduction- = RSum- | RCount- | RMin- | RMax- | RMean- | RStd- | RVar- | RTop2Sum- deriving (Eq, Show)--{- | Coerce an unboxed Int or Double column to an unboxed Double vector for the-moment/mean/sd/median family. Returns 'Nothing' for boxed, nullable, or other-element types (the caller then falls back to the interpreter).--}-scatterColumnToDouble :: Column -> Maybe (VU.Vector Double)-scatterColumnToDouble = \case- UnboxedColumn Nothing (v :: VU.Vector a) ->- case testEquality (typeRep @a) (typeRep @Double) of- Just Refl -> Just v- Nothing -> case testEquality (typeRep @a) (typeRep @Int) of- Just Refl -> Just (VU.map fromIntegral v)- Nothing -> Nothing- p@(PackedText _ _) -> scatterColumnToDouble (materializePacked p)- _ -> Nothing--scatterReduce ::- Reduction -> VU.Vector Int -> Int -> Column -> Maybe Column-scatterReduce red g nGroups col = case col of- UnboxedColumn Nothing (v :: VU.Vector a) ->- case testEquality (typeRep @a) (typeRep @Int) of- Just Refl -> Just (reduceTyped red g nGroups v intIdent)- Nothing -> case testEquality (typeRep @a) (typeRep @Double) of- Just Refl -> Just (reduceTyped red g nGroups v dblIdent)- Nothing -> Nothing- p@(PackedText _ _) -> scatterReduce red g nGroups (materializePacked p)- _ -> Nothing-{-# INLINEABLE scatterReduce #-}---- | Per-type seed identities for the order-preserving reductions.-data Idents a = Idents {minSeed :: !a, maxSeed :: !a}--intIdent :: Idents Int-intIdent = Idents maxBound minBound--dblIdent :: Idents Double-dblIdent = Idents (1 / 0) (negate (1 / 0))--reduceTyped ::- forall a.- (Columnable a, VU.Unbox a, Num a, Ord a, Real a) =>- Reduction -> VU.Vector Int -> Int -> VU.Vector a -> Idents a -> Column-reduceTyped red g nGroups v idents = case red of- RCount -> fromUnboxedVector (countScatter g nGroups)- RSum -> fromUnboxedVector (sumScatter g nGroups v)- RMin -> fromUnboxedVector (extremaScatter min (minSeed idents) g nGroups v)- RMax -> fromUnboxedVector (extremaScatter max (maxSeed idents) g nGroups v)- RMean -> fromUnboxedVector (meanScatter g nGroups v)- RVar -> fromUnboxedVector (varScatter False g nGroups v)- RStd -> fromUnboxedVector (varScatter True g nGroups v)- RTop2Sum -> fromUnboxedVector (top2Scatter g nGroups v)-{-# INLINE reduceTyped #-}--countScatter :: VU.Vector Int -> Int -> VU.Vector Int-countScatter g nGroups = runST $ do- cnt <- VUM.replicate nGroups (0 :: Int)- let n = VU.length g- go !i- | i >= n = pure ()- | otherwise = do- let !k = VU.unsafeIndex g i- c <- VUM.unsafeRead cnt k- VUM.unsafeWrite cnt k (c + 1)- go (i + 1)- go 0- VU.unsafeFreeze cnt--sumScatter ::- (VU.Unbox a, Num a) => VU.Vector Int -> Int -> VU.Vector a -> VU.Vector a-sumScatter g nGroups v = runST $ do- s <- VUM.replicate nGroups 0- let n = VU.length v- go !i- | i >= n = pure ()- | otherwise = do- let !k = VU.unsafeIndex g i- cur <- VUM.unsafeRead s k- VUM.unsafeWrite s k (cur + VU.unsafeIndex v i)- go (i + 1)- go 0- VU.unsafeFreeze s-{-# INLINE sumScatter #-}--extremaScatter ::- (VU.Unbox a) =>- (a -> a -> a) -> a -> VU.Vector Int -> Int -> VU.Vector a -> VU.Vector a-extremaScatter combine seed g nGroups v = runST $ do- m <- VUM.replicate nGroups seed- let n = VU.length v- go !i- | i >= n = pure ()- | otherwise = do- let !k = VU.unsafeIndex g i- cur <- VUM.unsafeRead m k- VUM.unsafeWrite m k (combine cur (VU.unsafeIndex v i))- go (i + 1)- go 0- VU.unsafeFreeze m-{-# INLINE extremaScatter #-}--meanScatter ::- (VU.Unbox a, Real a) => VU.Vector Int -> Int -> VU.Vector a -> VU.Vector Double-meanScatter g nGroups v = runST $ do- s <- VUM.replicate nGroups (0 :: Double)- cnt <- VUM.replicate nGroups (0 :: Int)- scatterSumCount g v s cnt- finalizeMean nGroups s cnt-{-# INLINE meanScatter #-}--scatterSumCount ::- (VU.Unbox a, Real a) =>- VU.Vector Int ->- VU.Vector a ->- VUM.MVector s Double ->- VUM.MVector s Int ->- ST s ()-scatterSumCount g v s cnt = go 0- where- n = VU.length v- go !i- | i >= n = pure ()- | otherwise = do- let !k = VU.unsafeIndex g i- !x = realToFrac (VU.unsafeIndex v i)- curS <- VUM.unsafeRead s k- VUM.unsafeWrite s k (curS + x)- curC <- VUM.unsafeRead cnt k- VUM.unsafeWrite cnt k (curC + 1)- go (i + 1)-{-# INLINE scatterSumCount #-}--finalizeMean ::- Int -> VUM.MVector s Double -> VUM.MVector s Int -> ST s (VU.Vector Double)-finalizeMean nGroups s cnt = do- out <- VUM.new nGroups- let go !k- | k >= nGroups = pure ()- | otherwise = do- sv <- VUM.unsafeRead s k- c <- VUM.unsafeRead cnt k- VUM.unsafeWrite out k (if c == 0 then 0 / 0 else sv / fromIntegral c)- go (k + 1)- go 0- VU.unsafeFreeze out--varScatter ::- (VU.Unbox a, Real a) =>- Bool -> VU.Vector Int -> Int -> VU.Vector a -> VU.Vector Double-varScatter takeSqrt g nGroups v = runST $ do- cnt <- VUM.replicate nGroups (0 :: Int)- meanV <- VUM.replicate nGroups (0 :: Double)- m2 <- VUM.replicate nGroups (0 :: Double)- let n = VU.length v- go !i- | i >= n = pure ()- | otherwise = do- let !k = VU.unsafeIndex g i- !x = realToFrac (VU.unsafeIndex v i)- c <- VUM.unsafeRead cnt k- mu <- VUM.unsafeRead meanV k- mm <- VUM.unsafeRead m2 k- let !c' = c + 1- !delta = x - mu- !mu' = mu + delta / fromIntegral c'- !mm' = mm + delta * (x - mu')- VUM.unsafeWrite cnt k c'- VUM.unsafeWrite meanV k mu'- VUM.unsafeWrite m2 k mm'- go (i + 1)- go 0- out <- VUM.new nGroups- let fin !k- | k >= nGroups = pure ()- | otherwise = do- c <- VUM.unsafeRead cnt k- mm <- VUM.unsafeRead m2 k- let var = if c < 2 then 0 else mm / fromIntegral (c - 1)- VUM.unsafeWrite out k (if takeSqrt then sqrt var else var)- fin (k + 1)- fin 0- VU.unsafeFreeze out-{-# INLINE varScatter #-}--top2Scatter ::- (VU.Unbox a, Real a) => VU.Vector Int -> Int -> VU.Vector a -> VU.Vector Double-top2Scatter g nGroups v = runST $ do- let ninf = negate (1 / 0) :: Double- m1 <- VUM.replicate nGroups ninf- m2 <- VUM.replicate nGroups ninf- let n = VU.length v- go !i- | i >= n = pure ()- | otherwise = do- let !k = VU.unsafeIndex g i- !x = realToFrac (VU.unsafeIndex v i)- a1 <- VUM.unsafeRead m1 k- if x > a1- then do- VUM.unsafeWrite m1 k x- VUM.unsafeWrite m2 k a1- else do- a2 <- VUM.unsafeRead m2 k- when (x > a2) (VUM.unsafeWrite m2 k x)- go (i + 1)- go 0- out <- VUM.new nGroups- let fin !k- | k >= nGroups = pure ()- | otherwise = do- a1 <- VUM.unsafeRead m1 k- a2 <- VUM.unsafeRead m2 k- let s = (if isInfinite a1 then 0 else a1) + (if isInfinite a2 then 0 else a2)- VUM.unsafeWrite out k s- fin (k + 1)- fin 0- VU.unsafeFreeze out-{-# INLINE top2Scatter #-}
− src-internal/DataFrame/Internal/AggKernelDirect.hs
@@ -1,338 +0,0 @@-{-# LANGUAGE BangPatterns #-}-{-# LANGUAGE ExplicitNamespaces #-}-{-# LANGUAGE FlexibleContexts #-}-{-# LANGUAGE GADTs #-}-{-# LANGUAGE ScopedTypeVariables #-}-{-# LANGUAGE TypeApplications #-}--module DataFrame.Internal.AggKernelDirect (- directThreshold,- directReduce,-) where--import Control.Concurrent (forkIO, getNumCapabilities)-import Control.Concurrent.MVar (newEmptyMVar, putMVar, takeMVar)-import Control.Exception (SomeException, throwIO, try)-import Data.Type.Equality (TestEquality (..), type (:~:) (Refl))-import qualified Data.Vector.Unboxed as VU-import qualified Data.Vector.Unboxed.Mutable as VUM-import System.IO.Unsafe (unsafePerformIO)-import Type.Reflection (typeRep)--import DataFrame.Internal.AggKernel (Reduction (..))-import DataFrame.Internal.Column (- Column (..),- fromUnboxedVector,- materializePacked,- )--{- | Group-domain size at or below which the direct-indexed accumulator path is-taken; wider domains keep the group-range kernel. The admitted reductions are-order-independent, so the per-worker accumulator merge is exact.--}-directThreshold :: Int-directThreshold = 262144--capabilities :: Int-capabilities = unsafePerformIO getNumCapabilities-{-# NOINLINE capabilities #-}--{- | Below this many rows the parallel fan-out is not worth it; a single-sequential direct pass runs instead (tiny accumulator, one tight loop). Matches-the grouping/scatter parallel threshold.--}-parThreshold :: Int-parThreshold = 200000--{- | Run a recognised reduction through the direct-indexed path. 'Nothing' (so-the caller falls back to the order-preserving kernel) unless the reduction is-order-independent at this element type AND the column is a clean unboxed Int/Double.--}-directReduce :: Reduction -> VU.Vector Int -> Int -> Column -> Maybe Column-directReduce red g nGroups col = case col of- UnboxedColumn Nothing (v :: VU.Vector a) ->- case testEquality (typeRep @a) (typeRep @Int) of- Just Refl -> directInt red g nGroups v- Nothing -> case testEquality (typeRep @a) (typeRep @Double) of- Just Refl -> directDouble red g nGroups v- Nothing -> Nothing- p@(PackedText _ _) -> directReduce red g nGroups (materializePacked p)- _ -> Nothing-{-# INLINEABLE directReduce #-}---- | The order-independent reductions over an Int column.-directInt :: Reduction -> VU.Vector Int -> Int -> VU.Vector Int -> Maybe Column-directInt red g nGroups v = case red of- RCount -> Just (fromUnboxedVector (countDirect g nGroups (VU.length v)))- RSum -> Just (fromUnboxedVector (sumIntDirect g nGroups v))- RMin -> Just (fromUnboxedVector (extremaIntDirect True g nGroups v))- RMax -> Just (fromUnboxedVector (extremaIntDirect False g nGroups v))- RMean -> Just (fromUnboxedVector (meanIntDirect g nGroups v))- _ -> Nothing--{- | Over a Double column only @count@ is order-independent; the float-sum/mean/variance reductions must keep the order-preserving kernel.--}-directDouble ::- Reduction -> VU.Vector Int -> Int -> VU.Vector Double -> Maybe Column-directDouble red g nGroups v = case red of- RCount -> Just (fromUnboxedVector (countDirect g nGroups (VU.length v)))- _ -> Nothing---- | Whether to fan out at this row count.-shouldPar :: Int -> Bool-shouldPar n = n >= parThreshold && capabilities > 1--{- | Fork @caps@ workers over disjoint contiguous row ranges of @[0, n)@, each-producing its own private accumulator (no shared array, no sync). Returns the-partials in worker order for the caller's merge; rethrows the first failure.--}-runPartialsOver ::- Int -> Int -> (Int -> Int -> IO (VUM.IOVector Int)) -> IO [VUM.IOVector Int]-runPartialsOver n caps fill = do- let !per = (n + caps - 1) `div` caps- spawn w = do- var <- newEmptyMVar- let !lo = min n (w * per)- !hi = min n (lo + per)- _ <- forkIO (try (fill lo hi) >>= putMVar var)- pure var- vars <- mapM spawn [0 .. caps - 1]- results <- mapM takeMVar vars- mapM (either (throwIO @SomeException) pure) results--{- | As 'runPartialsOver' but each worker produces a PAIR of accumulators (e.g.-sum and count for the fused integer mean).--}-runPartialsPairOver ::- Int ->- Int ->- (Int -> Int -> IO (VUM.IOVector Int, VUM.IOVector Int)) ->- IO [(VUM.IOVector Int, VUM.IOVector Int)]-runPartialsPairOver n caps fill = do- let !per = (n + caps - 1) `div` caps- spawn w = do- var <- newEmptyMVar- let !lo = min n (w * per)- !hi = min n (lo + per)- _ <- forkIO (try (fill lo hi) >>= putMVar var)- pure var- vars <- mapM spawn [0 .. caps - 1]- results <- mapM takeMVar vars- mapM (either (throwIO @SomeException) pure) results------------------------------------------------------------------------------------ Count (order-independent: per-group row count)----------------------------------------------------------------------------------countDirect :: VU.Vector Int -> Int -> Int -> VU.Vector Int-countDirect g nGroups n- | not (shouldPar n) =- unsafePerformIO (countChunk g nGroups 0 n >>= VU.unsafeFreeze)- | otherwise = unsafePerformIO $ do- parts <- runPartialsOver n capabilities (countChunk g nGroups)- mergeIntSum nGroups parts-{-# NOINLINE countDirect #-}--countChunk :: VU.Vector Int -> Int -> Int -> Int -> IO (VUM.IOVector Int)-countChunk g nGroups lo hi = do- acc <- VUM.replicate nGroups (0 :: Int)- let go !i- | i >= hi = pure ()- | otherwise = do- let !k = VU.unsafeIndex g i- c <- VUM.unsafeRead acc k- VUM.unsafeWrite acc k (c + 1)- go (i + 1)- go lo- pure acc------------------------------------------------------------------------------------ Integer sum (exact: merge order irrelevant)----------------------------------------------------------------------------------sumIntDirect :: VU.Vector Int -> Int -> VU.Vector Int -> VU.Vector Int-sumIntDirect g nGroups v- | not (shouldPar n) =- unsafePerformIO (sumIntChunk g v nGroups 0 n >>= VU.unsafeFreeze)- | otherwise = unsafePerformIO $ do- parts <- runPartialsOver n capabilities (sumIntChunk g v nGroups)- mergeIntSum nGroups parts- where- !n = VU.length v-{-# NOINLINE sumIntDirect #-}--sumIntChunk ::- VU.Vector Int -> VU.Vector Int -> Int -> Int -> Int -> IO (VUM.IOVector Int)-sumIntChunk g v nGroups lo hi = do- acc <- VUM.replicate nGroups (0 :: Int)- let go !i- | i >= hi = pure ()- | otherwise = do- let !k = VU.unsafeIndex g i- c <- VUM.unsafeRead acc k- VUM.unsafeWrite acc k (c + VU.unsafeIndex v i)- go (i + 1)- go lo- pure acc------------------------------------------------------------------------------------ Integer min / max (order-independent)----------------------------------------------------------------------------------extremaIntDirect ::- Bool -> VU.Vector Int -> Int -> VU.Vector Int -> VU.Vector Int-extremaIntDirect isMin g nGroups v- | not (shouldPar n) =- unsafePerformIO (extremaIntChunk isMin g v nGroups 0 n >>= VU.unsafeFreeze)- | otherwise = unsafePerformIO $ do- parts <- runPartialsOver n capabilities (extremaIntChunk isMin g v nGroups)- mergeExtremaInt isMin nGroups parts- where- !n = VU.length v-{-# NOINLINE extremaIntDirect #-}--extremaIntChunk ::- Bool ->- VU.Vector Int ->- VU.Vector Int ->- Int ->- Int ->- Int ->- IO (VUM.IOVector Int)-extremaIntChunk isMin g v nGroups lo hi = do- let !seed = if isMin then maxBound else minBound- combine a b = if isMin then min a b else max a b- acc <- VUM.replicate nGroups seed- let go !i- | i >= hi = pure ()- | otherwise = do- let !k = VU.unsafeIndex g i- c <- VUM.unsafeRead acc k- VUM.unsafeWrite acc k (combine c (VU.unsafeIndex v i))- go (i + 1)- go lo- pure acc------------------------------------------------------------------------------------ Integer mean (exact integer sum + count, divided once -> order-independent)----------------------------------------------------------------------------------{- | Integer mean in ONE fused pass: a running integer sum and count per group,-divided once at finalize. The integer sum is exact, so the parallel partial-merge is byte-identical to the sequential single pass at any @-N@.--}-meanIntDirect :: VU.Vector Int -> Int -> VU.Vector Int -> VU.Vector Double-meanIntDirect g nGroups v- | not (shouldPar n) = unsafePerformIO $ do- (s, c) <- meanIntChunk g v nGroups 0 n- finalizeMeanInt nGroups s c- | otherwise = unsafePerformIO $ do- parts <- runPartialsPairOver n capabilities (meanIntChunk g v nGroups)- (s, c) <- mergePair nGroups parts- finalizeMeanInt nGroups s c- where- !n = VU.length v-{-# NOINLINE meanIntDirect #-}--meanIntChunk ::- VU.Vector Int ->- VU.Vector Int ->- Int ->- Int ->- Int ->- IO (VUM.IOVector Int, VUM.IOVector Int)-meanIntChunk g v nGroups lo hi = do- s <- VUM.replicate nGroups (0 :: Int)- c <- VUM.replicate nGroups (0 :: Int)- let go !i- | i >= hi = pure ()- | otherwise = do- let !k = VU.unsafeIndex g i- sv <- VUM.unsafeRead s k- VUM.unsafeWrite s k (sv + VU.unsafeIndex v i)- cv <- VUM.unsafeRead c k- VUM.unsafeWrite c k (cv + 1)- go (i + 1)- go lo- pure (s, c)--finalizeMeanInt ::- Int -> VUM.IOVector Int -> VUM.IOVector Int -> IO (VU.Vector Double)-finalizeMeanInt nGroups s c = do- out <- VUM.new nGroups- let go !k- | k >= nGroups = pure ()- | otherwise = do- sv <- VUM.unsafeRead s k- cv <- VUM.unsafeRead c k- VUM.unsafeWrite- out- k- (if cv == 0 then 0 / 0 else fromIntegral sv / fromIntegral cv)- go (k + 1)- go 0- VU.unsafeFreeze out------------------------------------------------------------------------------------ Partial accumulation + merge----------------------------------------------------------------------------------mergeIntSum :: Int -> [VUM.IOVector Int] -> IO (VU.Vector Int)-mergeIntSum nGroups parts = case parts of- [] -> VU.unsafeFreeze =<< VUM.replicate nGroups 0- (p0 : rest) -> do- let add !p = do- let go !k- | k >= nGroups = pure ()- | otherwise = do- a <- VUM.unsafeRead p0 k- b <- VUM.unsafeRead p k- VUM.unsafeWrite p0 k (a + b)- go (k + 1)- go 0- mapM_ add rest- VU.unsafeFreeze p0--{- | Merge per-worker (sum, count) partials into the first worker's pair by-exact integer addition; returns the accumulated pair for finalize.--}-mergePair ::- Int ->- [(VUM.IOVector Int, VUM.IOVector Int)] ->- IO (VUM.IOVector Int, VUM.IOVector Int)-mergePair nGroups parts = case parts of- [] -> (,) <$> VUM.replicate nGroups 0 <*> VUM.replicate nGroups 0- ((s0, c0) : rest) -> do- let add (s, c) = do- let go !k- | k >= nGroups = pure ()- | otherwise = do- sa <- VUM.unsafeRead s0 k- sb <- VUM.unsafeRead s k- VUM.unsafeWrite s0 k (sa + sb)- ca <- VUM.unsafeRead c0 k- cb <- VUM.unsafeRead c k- VUM.unsafeWrite c0 k (ca + cb)- go (k + 1)- go 0- mapM_ add rest- pure (s0, c0)--mergeExtremaInt :: Bool -> Int -> [VUM.IOVector Int] -> IO (VU.Vector Int)-mergeExtremaInt isMin nGroups parts = case parts of- [] ->- VU.unsafeFreeze =<< VUM.replicate nGroups (if isMin then maxBound else minBound)- (p0 : rest) -> do- let combine a b = if isMin then min a b else max a b- add !p = do- let go !k- | k >= nGroups = pure ()- | otherwise = do- a <- VUM.unsafeRead p0 k- b <- VUM.unsafeRead p k- VUM.unsafeWrite p0 k (combine a b)- go (k + 1)- go 0- mapM_ add rest- VU.unsafeFreeze p0
− src-internal/DataFrame/Internal/AggKernelPar.hs
@@ -1,391 +0,0 @@-{-# LANGUAGE BangPatterns #-}-{-# LANGUAGE ExplicitNamespaces #-}-{-# LANGUAGE FlexibleContexts #-}-{-# LANGUAGE GADTs #-}-{-# LANGUAGE ScopedTypeVariables #-}-{-# LANGUAGE TypeApplications #-}---- | Parallel scatter-accumulate aggregation kernel.-module DataFrame.Internal.AggKernelPar (- scatterReducePar,- momentScatterPar,-) where--import Control.Concurrent (forkIO, getNumCapabilities)-import Control.Concurrent.MVar (newEmptyMVar, putMVar, takeMVar)-import Control.Exception (SomeException, throwIO, try)-import Control.Monad (when)-import Data.Type.Equality (TestEquality (..), type (:~:) (Refl))-import qualified Data.Vector.Unboxed as VU-import qualified Data.Vector.Unboxed.Mutable as VUM-import System.IO.Unsafe (unsafePerformIO)-import Type.Reflection (typeRep)--import DataFrame.Internal.AggKernel (- Reduction (..),- scatterColumnToDouble,- scatterReduce,- )-import DataFrame.Internal.AggPlan (Moments (..), momentScatter)-import DataFrame.Internal.Column (- Column (..),- Columnable,- fromUnboxedVector,- materializePacked,- )--parThreshold :: Int-parThreshold = 200000--capabilities :: Int-capabilities = unsafePerformIO getNumCapabilities-{-# NOINLINE capabilities #-}---- | Whether to take the parallel path at this row count.-shouldPar :: Int -> Bool-shouldPar n = n >= parThreshold && capabilities > 1--groupRangeBounds :: VU.Vector Int -> Int -> Int -> VU.Vector Int-groupRangeBounds offs nGroups caps = VU.create $ do- b <- VUM.new (caps + 1)- let !nRows = VU.unsafeIndex offs nGroups- !per = max 1 ((nRows + caps - 1) `div` caps)- adv !target !gg- | gg >= nGroups = nGroups- | VU.unsafeIndex offs gg >= target = gg- | otherwise = adv target (gg + 1)- go !w !prev- | w >= caps = VUM.unsafeWrite b caps nGroups- | otherwise = do- let !target = min nRows (w * per)- !g = adv target prev- VUM.unsafeWrite b w g- go (w + 1) g- VUM.unsafeWrite b 0 0- go 1 0- pure b--forEachRange :: VU.Vector Int -> Int -> (Int -> Int -> IO ()) -> IO ()-forEachRange bounds caps act- | caps <= 1 = act (VU.unsafeIndex bounds 0) (VU.unsafeIndex bounds caps)- | otherwise = do- vars <- mapM spawn [0 .. caps - 1]- results <- mapM takeMVar vars- mapM_ (either (throwIO :: SomeException -> IO ()) pure) results- where- spawn w = do- var <- newEmptyMVar- let !s = VU.unsafeIndex bounds w- !e = VU.unsafeIndex bounds (w + 1)- _ <- forkIO (try (act s e) >>= putMVar var)- pure var--scatterReducePar ::- Reduction -> VU.Vector Int -> VU.Vector Int -> Int -> Column -> Maybe Column-scatterReducePar red vis offs nGroups col- | not (shouldPar (VU.length vis)) || nGroups <= 1 =- scatterReduce red (rtgFromVis vis offs nGroups) nGroups col- | otherwise = case col of- UnboxedColumn Nothing (v :: VU.Vector a) ->- case testEquality (typeRep @a) (typeRep @Int) of- Just Refl -> Just (reduceParTyped red vis offs nGroups v intIdent)- Nothing -> case testEquality (typeRep @a) (typeRep @Double) of- Just Refl -> Just (reduceParTyped red vis offs nGroups v dblIdent)- Nothing -> Nothing- p@(PackedText _ _) -> scatterReducePar red vis offs nGroups (materializePacked p)- _ -> Nothing-{-# NOINLINE scatterReducePar #-}--rtgFromVis :: VU.Vector Int -> VU.Vector Int -> Int -> VU.Vector Int-rtgFromVis vis offs nGroups = VU.create $ do- let n = VU.length vis- rtg <- VUM.new (max 1 n)- let go !g- | g >= nGroups = pure ()- | otherwise = do- let !e = VU.unsafeIndex offs (g + 1)- inner !pos- | pos >= e = pure ()- | otherwise = do- VUM.unsafeWrite rtg (VU.unsafeIndex vis pos) g- inner (pos + 1)- inner (VU.unsafeIndex offs g)- go (g + 1)- go 0- pure rtg--data Idents a = Idents {minSeed :: !a, maxSeed :: !a}--intIdent :: Idents Int-intIdent = Idents maxBound minBound--dblIdent :: Idents Double-dblIdent = Idents (1 / 0) (negate (1 / 0))--reduceParTyped ::- forall a.- (Columnable a, VU.Unbox a, Num a, Ord a, Real a) =>- Reduction ->- VU.Vector Int ->- VU.Vector Int ->- Int ->- VU.Vector a ->- Idents a ->- Column-reduceParTyped red vis offs nGroups v idents =- let !caps = capabilities- !bounds = groupRangeBounds offs nGroups caps- in case red of- RCount -> fromUnboxedVector (unsafePerformIO (countPar vis offs nGroups caps bounds))- RSum -> fromUnboxedVector (unsafePerformIO (sumPar vis offs nGroups v caps bounds))- RMin ->- fromUnboxedVector- (unsafePerformIO (extremaPar min (minSeed idents) vis offs nGroups v caps bounds))- RMax ->- fromUnboxedVector- (unsafePerformIO (extremaPar max (maxSeed idents) vis offs nGroups v caps bounds))- RMean -> fromUnboxedVector (unsafePerformIO (meanPar vis offs nGroups v caps bounds))- RVar ->- fromUnboxedVector- (unsafePerformIO (varPar False vis offs nGroups v caps bounds))- RStd ->- fromUnboxedVector (unsafePerformIO (varPar True vis offs nGroups v caps bounds))- RTop2Sum -> fromUnboxedVector (unsafePerformIO (top2Par vis offs nGroups v caps bounds))-{-# INLINE reduceParTyped #-}---- | Iterate the rows of groups @[gs, ge)@ in @valueIndices@/group order.-overGroups ::- VU.Vector Int -> VU.Vector Int -> Int -> Int -> (Int -> Int -> IO ()) -> IO ()-overGroups vis offs gs ge step = grp gs- where- grp !g- | g >= ge = pure ()- | otherwise = do- let !e = VU.unsafeIndex offs (g + 1)- inner !pos- | pos >= e = pure ()- | otherwise = step g (VU.unsafeIndex vis pos) >> inner (pos + 1)- inner (VU.unsafeIndex offs g)- grp (g + 1)-{-# INLINE overGroups #-}--countPar ::- VU.Vector Int ->- VU.Vector Int ->- Int ->- Int ->- VU.Vector Int ->- IO (VU.Vector Int)-countPar _vis offs nGroups caps bounds = do- out <- VUM.replicate nGroups (0 :: Int)- forEachRange bounds caps $ \gs ge ->- let grp !g- | g >= ge = pure ()- | otherwise = do- let !c = VU.unsafeIndex offs (g + 1) - VU.unsafeIndex offs g- VUM.unsafeWrite out g c- grp (g + 1)- in grp gs- VU.unsafeFreeze out--sumPar ::- (VU.Unbox a, Num a) =>- VU.Vector Int ->- VU.Vector Int ->- Int ->- VU.Vector a ->- Int ->- VU.Vector Int ->- IO (VU.Vector a)-sumPar vis offs nGroups v caps bounds = do- out <- VUM.replicate nGroups 0- forEachRange bounds caps $ \gs ge ->- overGroups vis offs gs ge $ \g row -> do- cur <- VUM.unsafeRead out g- VUM.unsafeWrite out g (cur + VU.unsafeIndex v row)- VU.unsafeFreeze out-{-# INLINE sumPar #-}--extremaPar ::- (VU.Unbox a) =>- (a -> a -> a) ->- a ->- VU.Vector Int ->- VU.Vector Int ->- Int ->- VU.Vector a ->- Int ->- VU.Vector Int ->- IO (VU.Vector a)-extremaPar combine seed vis offs nGroups v caps bounds = do- out <- VUM.replicate nGroups seed- forEachRange bounds caps $ \gs ge ->- overGroups vis offs gs ge $ \g row -> do- cur <- VUM.unsafeRead out g- VUM.unsafeWrite out g (combine cur (VU.unsafeIndex v row))- VU.unsafeFreeze out-{-# INLINE extremaPar #-}--meanPar ::- (VU.Unbox a, Real a) =>- VU.Vector Int ->- VU.Vector Int ->- Int ->- VU.Vector a ->- Int ->- VU.Vector Int ->- IO (VU.Vector Double)-meanPar vis offs nGroups v caps bounds = do- s <- VUM.replicate nGroups (0 :: Double)- cnt <- VUM.replicate nGroups (0 :: Int)- forEachRange bounds caps $ \gs ge ->- overGroups vis offs gs ge $ \g row -> do- let !x = realToFrac (VU.unsafeIndex v row)- cs <- VUM.unsafeRead s g- VUM.unsafeWrite s g (cs + x)- cc <- VUM.unsafeRead cnt g- VUM.unsafeWrite cnt g (cc + 1)- out <- VUM.new nGroups- let fin !k- | k >= nGroups = pure ()- | otherwise = do- sv <- VUM.unsafeRead s k- c <- VUM.unsafeRead cnt k- VUM.unsafeWrite out k (if c == 0 then 0 / 0 else sv / fromIntegral c)- fin (k + 1)- fin 0- VU.unsafeFreeze out-{-# INLINE meanPar #-}--varPar ::- (VU.Unbox a, Real a) =>- Bool ->- VU.Vector Int ->- VU.Vector Int ->- Int ->- VU.Vector a ->- Int ->- VU.Vector Int ->- IO (VU.Vector Double)-varPar takeSqrt vis offs nGroups v caps bounds = do- cnt <- VUM.replicate nGroups (0 :: Int)- meanV <- VUM.replicate nGroups (0 :: Double)- m2 <- VUM.replicate nGroups (0 :: Double)- forEachRange bounds caps $ \gs ge ->- overGroups vis offs gs ge $ \g row -> do- let !x = realToFrac (VU.unsafeIndex v row)- c <- VUM.unsafeRead cnt g- mu <- VUM.unsafeRead meanV g- mm <- VUM.unsafeRead m2 g- let !c' = c + 1- !delta = x - mu- !mu' = mu + delta / fromIntegral c'- !mm' = mm + delta * (x - mu')- VUM.unsafeWrite cnt g c'- VUM.unsafeWrite meanV g mu'- VUM.unsafeWrite m2 g mm'- out <- VUM.new nGroups- let fin !k- | k >= nGroups = pure ()- | otherwise = do- c <- VUM.unsafeRead cnt k- mm <- VUM.unsafeRead m2 k- let var = if c < 2 then 0 else mm / fromIntegral (c - 1)- VUM.unsafeWrite out k (if takeSqrt then sqrt var else var)- fin (k + 1)- fin 0- VU.unsafeFreeze out-{-# INLINE varPar #-}--top2Par ::- (VU.Unbox a, Real a) =>- VU.Vector Int ->- VU.Vector Int ->- Int ->- VU.Vector a ->- Int ->- VU.Vector Int ->- IO (VU.Vector Double)-top2Par vis offs nGroups v caps bounds = do- let ninf = negate (1 / 0) :: Double- m1 <- VUM.replicate nGroups ninf- m2 <- VUM.replicate nGroups ninf- forEachRange bounds caps $ \gs ge ->- overGroups vis offs gs ge $ \g row -> do- let !x = realToFrac (VU.unsafeIndex v row)- a1 <- VUM.unsafeRead m1 g- if x > a1- then do- VUM.unsafeWrite m1 g x- VUM.unsafeWrite m2 g a1- else do- a2 <- VUM.unsafeRead m2 g- when (x > a2) (VUM.unsafeWrite m2 g x)- out <- VUM.new nGroups- let fin !k- | k >= nGroups = pure ()- | otherwise = do- a1 <- VUM.unsafeRead m1 k- a2 <- VUM.unsafeRead m2 k- let sm = (if isInfinite a1 then 0 else a1) + (if isInfinite a2 then 0 else a2)- VUM.unsafeWrite out k sm- fin (k + 1)- fin 0- VU.unsafeFreeze out-{-# INLINE top2Par #-}------------------------------------------------------------------------------------ Parallel fused two-column moments (Q9)----------------------------------------------------------------------------------{- | Parallel counterpart of 'momentScatter': one fused pass over both columns,-each group's six sums accumulated within one worker's range. Byte-identical to-'momentScatter'. 'Nothing' unless both columns are non-null unboxed Int/Double.--}-momentScatterPar ::- VU.Vector Int -> VU.Vector Int -> Int -> Column -> Column -> Maybe Moments-momentScatterPar vis offs nGroups colX colY- | not (shouldPar (VU.length vis)) || nGroups <= 1 =- momentScatter (rtgFromVis vis offs nGroups) nGroups colX colY- | otherwise = do- xs <- scatterColumnToDouble colX- ys <- scatterColumnToDouble colY- let !caps = capabilities- !bounds = groupRangeBounds offs nGroups caps- pure (unsafePerformIO (momentPar vis offs nGroups xs ys caps bounds))-{-# NOINLINE momentScatterPar #-}--momentPar ::- VU.Vector Int ->- VU.Vector Int ->- Int ->- VU.Vector Double ->- VU.Vector Double ->- Int ->- VU.Vector Int ->- IO Moments-momentPar vis offs nGroups xs ys caps bounds = do- cnt <- VUM.replicate nGroups (0 :: Int)- sx <- VUM.replicate nGroups (0 :: Double)- sy <- VUM.replicate nGroups (0 :: Double)- sxx <- VUM.replicate nGroups (0 :: Double)- syy <- VUM.replicate nGroups (0 :: Double)- sxy <- VUM.replicate nGroups (0 :: Double)- let bump arr g d = VUM.unsafeRead arr g >>= \c -> VUM.unsafeWrite arr g (c + d)- forEachRange bounds caps $ \gs ge ->- overGroups vis offs gs ge $ \g row -> do- let !x = VU.unsafeIndex xs row- !y = VU.unsafeIndex ys row- VUM.unsafeRead cnt g >>= \c -> VUM.unsafeWrite cnt g (c + 1)- bump sx g x- bump sy g y- bump sxx g (x * x)- bump syy g (y * y)- bump sxy g (x * y)- Moments . fromUnboxedVector- <$> VU.unsafeFreeze cnt- <*> (fromUnboxedVector <$> VU.unsafeFreeze sx)- <*> (fromUnboxedVector <$> VU.unsafeFreeze sy)- <*> (fromUnboxedVector <$> VU.unsafeFreeze sxx)- <*> (fromUnboxedVector <$> VU.unsafeFreeze syy)- <*> (fromUnboxedVector <$> VU.unsafeFreeze sxy)
− src-internal/DataFrame/Internal/AggPlan.hs
@@ -1,299 +0,0 @@-{-# LANGUAGE AllowAmbiguousTypes #-}-{-# LANGUAGE BangPatterns #-}-{-# LANGUAGE ExplicitNamespaces #-}-{-# LANGUAGE GADTs #-}-{-# LANGUAGE LambdaCase #-}-{-# LANGUAGE OverloadedStrings #-}-{-# LANGUAGE ScopedTypeVariables #-}-{-# LANGUAGE TypeApplications #-}--{- | The aggregation fast-path planner and the two-column moment scatter.-'planAgg' recognises a supported aggregate shape over a clean unboxed Int/Double-column and returns an 'AggPlan'; 'momentScatter' fuses the six regression sums.--}-module DataFrame.Internal.AggPlan (- AggPlan (..),- planAgg,- Moments (..),- momentScatter,- MomentPlan (..),- planMoments,-) where--import qualified Data.Map.Strict as M-import qualified Data.Text as T-import Data.Type.Equality (TestEquality (..), type (:~:) (Refl))-import qualified Data.Vector.Unboxed as VU-import qualified Data.Vector.Unboxed.Mutable as VUM--import Control.Monad.ST (runST)-import DataFrame.Internal.AggKernel (Reduction (..), scatterColumnToDouble)-import DataFrame.Internal.Column (Column (..), fromUnboxedVector)-import DataFrame.Internal.DataFrame (- DataFrame (derivingExpressions),- GroupedDataFrame (..),- getColumn,- )-import DataFrame.Internal.Expression (- AggStrategy (..),- BinaryOp (binaryCommutative, binaryName),- Expr (..),- UExpr (..),- )-import Type.Reflection (Typeable, typeRep)--{- | The plan 'planAgg' produces for a recognised output expression. The median-plan carries only the column name (the holistic grouped sort lives in the-operations layer, where @vector-algorithms@ is available).--}-data AggPlan- = -- | A single scatter reduction over one named column.- PlanScatter Reduction T.Text- | -- | @max a - min b@ (Q7): two scatters then a vectorized combine.- PlanMaxMinusMin T.Text T.Text- | -- | Holistic median over one named column.- PlanMedian T.Text--{- | Inspect a named output expression; return @Just plan@ on a recognised shape-over a present clean column, else 'Nothing'. Nullable or non-Int/Double columns-are rejected here so the scatter only sees a clean unboxed vector.--}-planAgg :: GroupedDataFrame -> UExpr -> Maybe AggPlan-planAgg gdf (UExpr (expr :: Expr a)) = case expr of- Agg (FoldAgg tag _ _) (Col name) -> foldPlan tag name- Agg (MergeAgg tag _ _ _ _) (Col name) -> mergePlan tag name- Agg (CollectAgg tag _) (Col name) -> collectPlan tag name- Binary- op- (Agg (FoldAgg lt Nothing _) (Col a))- (Agg (FoldAgg rt Nothing _) (Col b)) ->- if binaryName op == "sub" && lt == "maximum" && rt == "minimum"- then requireBoth a b (PlanMaxMinusMin a b)- else Nothing- _ -> Nothing- where- foldPlan tag name = case tag of- "sum" -> require name (PlanScatter RSum name)- "minimum" -> require name (PlanScatter RMin name)- "maximum" -> require name (PlanScatter RMax name)- _ -> Nothing- mergePlan tag name = case tag of- "mean" -> outputType @Double >> require name (PlanScatter RMean name)- "count" -> outputType @Int >> require name (PlanScatter RCount name)- _ -> Nothing- outputType :: forall t. (Typeable t) => Maybe ()- outputType = case testEquality (typeRep @a) (typeRep @t) of- Just Refl -> Just ()- Nothing -> Nothing- collectPlan tag name = case tag of- "stddev" -> require name (PlanScatter RStd name)- "variance" -> require name (PlanScatter RVar name)- "top2Sum" -> require name (PlanScatter RTop2Sum name)- "median" -> require name (PlanMedian name)- _ -> Nothing- require name plan = colUnboxedNumeric name >> Just plan- requireBoth a b plan = colUnboxedNumeric a >> colUnboxedNumeric b >> Just plan- colUnboxedNumeric name = case getColumn name (fullDataframe gdf) of- Just c | isUnboxedNumeric c -> Just ()- _ -> Nothing---- | The matcher only fires on non-null unboxed Int/Double columns.-isUnboxedNumeric :: Column -> Bool-isUnboxedNumeric = \case- UnboxedColumn Nothing (_ :: VU.Vector a) ->- case testEquality (typeRep @a) (typeRep @Int) of- Just Refl -> True- Nothing -> case testEquality (typeRep @a) (typeRep @Double) of- Just Refl -> True- Nothing -> False- _ -> False--{- | A recognised moment (Q9 regression) aggregate group: six output columns that-form the sufficient statistics of two base columns @x@ and @y@. The caller runs-'momentScatter' once and binds each output name to a field of the result.--}-data MomentPlan = MomentPlan- { mpColX :: T.Text- , mpColY :: T.Text- , mpNName :: T.Text- , mpSxName :: T.Text- , mpSyName :: T.Text- , mpSxxName :: T.Text- , mpSyyName :: T.Text- , mpSxyName :: T.Text- }--{- | The shape of a sum's argument once unary coercions are peeled and derived-columns are resolved through @derivingExpressions@: either linear in one base-column or the product of two base columns (sorted).--}-data Term- = Lin T.Text- | Prod T.Text T.Text- deriving (Eq, Ord, Show)--{- | Recognise the moment shape across a whole @aggregate@ list: exactly-@count@, @sum(x)@, @sum(y)@, @sum(x*x)@, @sum(y*y)@, @sum(x*y)@ over two distinct-clean unboxed base columns. 'Nothing' on any other set.--}-planMoments :: GroupedDataFrame -> [(T.Text, UExpr)] -> Maybe MomentPlan-planMoments gdf aggs- | length aggs /= 6 = Nothing- | otherwise = do- let exprs = derivingExpressions (fullDataframe gdf)- roles <- traverse (classify exprs) aggs- let names = M.fromList [(r, nm) | (nm, r) <- roles]- nName <- M.lookup RoleN names- (x, y) <- pickBaseColumns roles- sxName <- M.lookup (RoleLin x) names- syName <- M.lookup (RoleLin y) names- sxxName <- M.lookup (RoleProd x x) names- syyName <- M.lookup (RoleProd y y) names- sxyName <- M.lookup (RoleProd x y) names- _ <- if x /= y then Just () else Nothing- _ <- colUnboxedNumeric x- _ <- colUnboxedNumeric y- pure- MomentPlan- { mpColX = x- , mpColY = y- , mpNName = nName- , mpSxName = sxName- , mpSyName = syName- , mpSxxName = sxxName- , mpSyyName = syyName- , mpSxyName = sxyName- }- where- colUnboxedNumeric name = case getColumn name (fullDataframe gdf) of- Just c | isUnboxedNumeric c -> Just ()- _ -> Nothing---- | The output role each named aggregation plays in the moment shape.-data Role- = RoleN- | RoleLin T.Text- | RoleProd T.Text T.Text- deriving (Eq, Ord, Show)---- | Tag a single named aggregation with its moment role, or reject the group.-classify :: M.Map T.Text UExpr -> (T.Text, UExpr) -> Maybe (T.Text, Role)-classify exprs (name, UExpr expr) = case expr of- Agg (MergeAgg "count" _ _ _ _) _ -> Just (name, RoleN)- Agg (FoldAgg "sum" _ _) arg -> (\t -> (name, termRole t)) <$> resolveTerm exprs (UExpr arg)- _ -> Nothing--termRole :: Term -> Role-termRole (Lin a) = RoleLin a-termRole (Prod a b) = RoleProd a b--{- | Resolve a (sum-argument) expression to its 'Term'. Peels @toDouble@-style-unary coercions, follows a derived column to its stored expression, and-recognises a commutative product of two linear terms.--}-resolveTerm :: M.Map T.Text UExpr -> UExpr -> Maybe Term-resolveTerm exprs = go (8 :: Int)- where- go 0 _ = Nothing- go fuel (UExpr e) = case e of- Col nm -> case M.lookup nm exprs of- Just ue -> go (fuel - 1) ue- Nothing -> Just (Lin nm)- Unary _ inner -> go (fuel - 1) (UExpr inner)- Binary op l r- | binaryName op == "mult" && binaryCommutative op -> do- Lin a <- go (fuel - 1) (UExpr l)- Lin b <- go (fuel - 1) (UExpr r)- Just (sortProd a b)- _ -> Nothing---- | Products are unordered: store the pair sorted so @x*y@ and @y*x@ unify.-sortProd :: T.Text -> T.Text -> Term-sortProd a b- | a <= b = Prod a b- | otherwise = Prod b a--{- | From the classified roles, find the unordered pair of base columns that the-linear sums name. There must be exactly two distinct linear-sum columns.--}-pickBaseColumns :: [(T.Text, Role)] -> Maybe (T.Text, T.Text)-pickBaseColumns roles =- case lins of- [a, b] | a /= b -> Just (a, b)- _ -> Nothing- where- lins = M.keys (M.fromList [(c, ()) | (_, RoleLin c) <- roles])--{- | The additive moment sums of two columns, each an @nGroups@-length column:-@(n, Sx, Sy, Sxx, Syy, Sxy)@.--}-data Moments = Moments- { mN :: Column- , mSx :: Column- , mSy :: Column- , mSxx :: Column- , mSyy :: Column- , mSxy :: Column- }--{- | One pass over two Double-coercible columns @x@ and @y@ filling the count and-five sums, collapsing the Q9 regression family's six folds into a single pass.-'Nothing' unless both columns are non-null unboxed Int/Double.--}-momentScatter :: VU.Vector Int -> Int -> Column -> Column -> Maybe Moments-momentScatter g nGroups colX colY = do- xs <- scatterColumnToDouble colX- ys <- scatterColumnToDouble colY- let (cnt, sx, sy, sxx, syy, sxy) = momentPass g nGroups xs ys- pure- Moments- { mN = fromUnboxedVector cnt- , mSx = fromUnboxedVector sx- , mSy = fromUnboxedVector sy- , mSxx = fromUnboxedVector sxx- , mSyy = fromUnboxedVector syy- , mSxy = fromUnboxedVector sxy- }--momentPass ::- VU.Vector Int ->- Int ->- VU.Vector Double ->- VU.Vector Double ->- ( VU.Vector Int- , VU.Vector Double- , VU.Vector Double- , VU.Vector Double- , VU.Vector Double- , VU.Vector Double- )-momentPass g nGroups xs ys = runST $ do- cnt <- VUM.replicate nGroups (0 :: Int)- sx <- VUM.replicate nGroups (0 :: Double)- sy <- VUM.replicate nGroups (0 :: Double)- sxx <- VUM.replicate nGroups (0 :: Double)- syy <- VUM.replicate nGroups (0 :: Double)- sxy <- VUM.replicate nGroups (0 :: Double)- let n = VU.length xs- bump arr k d = VUM.unsafeRead arr k >>= \c -> VUM.unsafeWrite arr k (c + d)- go !i- | i >= n = pure ()- | otherwise = do- let !k = VU.unsafeIndex g i- !x = VU.unsafeIndex xs i- !y = VU.unsafeIndex ys i- VUM.unsafeRead cnt k >>= \c -> VUM.unsafeWrite cnt k (c + 1)- bump sx k x- bump sy k y- bump sxx k (x * x)- bump syy k (y * y)- bump sxy k (x * y)- go (i + 1)- go 0- (,,,,,)- <$> VU.unsafeFreeze cnt- <*> VU.unsafeFreeze sx- <*> VU.unsafeFreeze sy- <*> VU.unsafeFreeze sxx- <*> VU.unsafeFreeze syy- <*> VU.unsafeFreeze sxy
+ src-internal/DataFrame/Internal/Aggregation/Kernel/Dense.hs view
@@ -0,0 +1,832 @@+{-# LANGUAGE BangPatterns #-}+{-# LANGUAGE ExplicitNamespaces #-}+{-# LANGUAGE FlexibleContexts #-}+{-# LANGUAGE GADTs #-}+{-# LANGUAGE ScopedTypeVariables #-}+{-# LANGUAGE TypeApplications #-}++{- | The low-cardinality DENSE reduction kernel: when the group domain is small,+the grouping layer's @rowToGroup@ already maps row -> group, so the reduction+scatters straight off it with no @valueIndices@ gather.++Parallel by ROW range with a private per-worker accumulator of @nGroups@ slots,+merged afterwards — which is why it needs a small domain, and why it admits only+order-independent reductions: the merge must be exact for the result to stay+byte-identical to @-N1@. Anything it rejects falls back to+"DataFrame.Internal.Aggregation.Kernel.Scatter".++The caller decides whether the domain is small enough; see @denseThreshold@ in+the operations layer.+-}+module DataFrame.Internal.Aggregation.Kernel.Dense (+ denseReduce,+ denseMaxMinusMin,+) where++import Control.Monad (when)+import Data.Type.Equality (TestEquality (..), type (:~:) (Refl))+import qualified Data.Vector.Unboxed as VU+import qualified Data.Vector.Unboxed.Mutable as VUM+import System.IO.Unsafe (unsafePerformIO)+import Type.Reflection (typeRep)++import DataFrame.Internal.Aggregation.Reduction (Reduction (..))+import DataFrame.Internal.Column (+ Column (..),+ fromUnboxedVector,+ materializePacked,+ )+import DataFrame.Internal.Control.Concurrent (+ capabilities,+ parThreshold,+ parallelChunks,+ shouldParallelize,+ )++{- | Group count at or below which the float (Double sum/mean) direct+reductions run as ONE sequential row-order pass: the accumulators fit in L2, the+pass is memory-bandwidth bound, and adding each value to its group in ascending+row order is exactly the order the group-range gather kernel uses (the grouping+layer's @valueIndices@ is a stable counting sort), so the result is+byte-identical to it. Above this the parallel chunked variant runs instead (see+'sumDblDense'). Var/std are never taken directly: any chunked merge of+variance state changes the float recurrence, and the group-range gather kernel+already runs them in parallel while replaying the interpreter's per-group+update order bit-for-bit.+-}+seqFloatGroups :: Int+seqFloatGroups = 65536++{- | Run a recognised reduction through the direct-indexed path. 'Nothing' (so+the caller falls back to the order-preserving kernel) unless the reduction is+admitted at this element type AND the column is a clean unboxed Int/Double.+-}+denseReduce :: Reduction -> VU.Vector Int -> Int -> Column -> Maybe Column+denseReduce red g nGroups col = case col of+ UnboxedColumn Nothing (v :: VU.Vector a) ->+ case testEquality (typeRep @a) (typeRep @Int) of+ Just Refl -> denseInt red g nGroups v+ Nothing -> case testEquality (typeRep @a) (typeRep @Double) of+ Just Refl -> denseDouble red g nGroups v+ Nothing -> Nothing+ p@(PackedText _ _) -> denseReduce red g nGroups (materializePacked p)+ _ -> Nothing+{-# INLINEABLE denseReduce #-}++{- | The reductions admitted over an Int column. Sum/min/max/mean/count are+exact in the Int domain (any merge order gives the same bits); top2sum selects+the two largest values (order-independent as a multiset selection) and only+adds them once at finalize. Var/std stay with the group-range gather kernel:+its per-group Welford recurrence replays the interpreter's update order+bit-for-bit, which no chunk-merged direct pass can.+-}+denseInt :: Reduction -> VU.Vector Int -> Int -> VU.Vector Int -> Maybe Column+denseInt red g nGroups v = case red of+ RCount -> Just (fromUnboxedVector (countDense g nGroups (VU.length v)))+ RSum -> Just (fromUnboxedVector (sumIntDense g nGroups v))+ RMin -> Just (fromUnboxedVector (extremaIntDense True g nGroups v))+ RMax -> Just (fromUnboxedVector (extremaIntDense False g nGroups v))+ RMean -> Just (fromUnboxedVector (meanIntDense g nGroups v))+ RTop2Sum -> Just (fromUnboxedVector (top2Dense g nGroups v))+ RTop2Snd -> Just (fromUnboxedVector (top2SndDense g nGroups v))+ _ -> Nothing++{- | The reductions admitted over a Double column. Count/min/max/top2sum are+order-independent (exact per-worker merge, byte-identical at any @-N@). The+float sum/mean run sequentially in row order below 'seqFloatGroups' (matching+the gather kernel's per-group addition order exactly) and as deterministic+chunked partials above it. Var/std keep the gather kernel (see 'denseInt').+-}+denseDouble ::+ Reduction -> VU.Vector Int -> Int -> VU.Vector Double -> Maybe Column+denseDouble red g nGroups v = case red of+ RCount -> Just (fromUnboxedVector (countDense g nGroups (VU.length v)))+ RSum -> Just (fromUnboxedVector (sumDblDense g nGroups v))+ RMean -> Just (fromUnboxedVector (meanDblDense g nGroups v))+ RMin -> Just (fromUnboxedVector (extremaDblDense True g nGroups v))+ RMax -> Just (fromUnboxedVector (extremaDblDense False g nGroups v))+ RTop2Sum -> Just (fromUnboxedVector (top2Dense g nGroups v))+ RTop2Snd -> Just (fromUnboxedVector (top2SndDense g nGroups v))+ _ -> Nothing++{- | The fused @max a - min b@ direct pass: BOTH extrema accumulate in one+streaming loop over the rows (min/max are order-independent, so the per-worker+merge is exact and the result byte-identical to the two gather passes it+replaces). 'Nothing' unless both columns are clean unboxed and same-typed+(Int/Int keeps the Int result of the interpreter; Double/Double the Double one);+mixed pairs keep the gather fallback.+-}+denseMaxMinusMin :: VU.Vector Int -> Int -> Column -> Column -> Maybe Column+denseMaxMinusMin g nGroups ca cb = case (ca, cb) of+ ( UnboxedColumn Nothing (va :: VU.Vector x)+ , UnboxedColumn Nothing (vb :: VU.Vector y)+ )+ | Just Refl <- testEquality (typeRep @x) (typeRep @Int)+ , Just Refl <- testEquality (typeRep @y) (typeRep @Int) ->+ Just (fromUnboxedVector (maxMinusMinDenseInt g nGroups va vb))+ | Just Refl <- testEquality (typeRep @x) (typeRep @Double)+ , Just Refl <- testEquality (typeRep @y) (typeRep @Double) ->+ Just (fromUnboxedVector (maxMinusMinDenseDbl g nGroups va vb))+ _ -> Nothing+{-# INLINEABLE denseMaxMinusMin #-}++{- | Monomorphic entry points: the 'testEquality' dispatch above only yields an+unsafe coercion, so a direct call to the polymorphic 'maxMinusMinDense' there+would stay at the abstract element type and never meet its SPECIALIZE rules+(measured ~3x on the whole pass); calling through these fixed-type wrappers+(the coercion lands on the argument) does.+-}+maxMinusMinDenseInt ::+ VU.Vector Int -> Int -> VU.Vector Int -> VU.Vector Int -> VU.Vector Int+maxMinusMinDenseInt = maxMinusMinDense minBound maxBound+{-# NOINLINE maxMinusMinDenseInt #-}++maxMinusMinDenseDbl ::+ VU.Vector Int -> Int -> VU.Vector Double -> VU.Vector Double -> VU.Vector Double+maxMinusMinDenseDbl = maxMinusMinDense (negate (1 / 0)) (1 / 0)+{-# NOINLINE maxMinusMinDenseDbl #-}++-- | Whether to fan out at this row count.+shouldPar :: Int -> Bool+shouldPar = shouldParallelize parThreshold++{- | Fork @caps@ workers over disjoint contiguous row ranges of @[0, n)@, each+producing its own private accumulator (no shared array, no sync). Returns the+partials in worker order for the caller's merge; rethrows the first failure.+The chunking is a fixed function of @n@ and @caps@, so any merge over the+partials is deterministic at a given @-N@.+-}+runPartialsOver ::+ Int -> Int -> (Int -> Int -> IO acc) -> IO [acc]+runPartialsOver n _caps = parallelChunks parThreshold n++-------------------------------------------------------------------------------+-- Count (order-independent: per-group row count)+-------------------------------------------------------------------------------++countDense :: VU.Vector Int -> Int -> Int -> VU.Vector Int+countDense g nGroups n = unsafePerformIO $ do+ parts <- runPartialsOver n capabilities (countChunk g nGroups)+ mergeIntSum nGroups parts+{-# NOINLINE countDense #-}++countChunk :: VU.Vector Int -> Int -> Int -> Int -> IO (VUM.IOVector Int)+countChunk g nGroups lo hi = do+ acc <- VUM.replicate nGroups (0 :: Int)+ let go !i+ | i >= hi = pure ()+ | otherwise = do+ let !k = VU.unsafeIndex g i+ c <- VUM.unsafeRead acc k+ VUM.unsafeWrite acc k (c + 1)+ go (i + 1)+ go lo+ pure acc++-------------------------------------------------------------------------------+-- Integer sum (exact: merge order irrelevant)+-------------------------------------------------------------------------------++sumIntDense :: VU.Vector Int -> Int -> VU.Vector Int -> VU.Vector Int+sumIntDense g nGroups v = unsafePerformIO $ do+ parts <- runPartialsOver (VU.length v) capabilities (sumIntChunk g v nGroups)+ mergeIntSum nGroups parts+{-# NOINLINE sumIntDense #-}++sumIntChunk ::+ VU.Vector Int -> VU.Vector Int -> Int -> Int -> Int -> IO (VUM.IOVector Int)+sumIntChunk g v nGroups lo hi = do+ acc <- VUM.replicate nGroups (0 :: Int)+ let go !i+ | i >= hi = pure ()+ | otherwise = do+ let !k = VU.unsafeIndex g i+ c <- VUM.unsafeRead acc k+ VUM.unsafeWrite acc k (c + VU.unsafeIndex v i)+ go (i + 1)+ go lo+ pure acc++-------------------------------------------------------------------------------+-- Double sum / mean (streaming scatter; chunked partials above seqFloatGroups)+-------------------------------------------------------------------------------++{- | Double group sums. At or below 'seqFloatGroups' a single sequential pass in+ascending row order — each group's additions happen in exactly the order the+group-range gather kernel performs them, so the result is byte-identical to it.+Above that, per-worker chunk partials merged in worker order: still+deterministic at a fixed @-N@, but the float summation order differs from the+sequential pass.+-}+sumDblDense :: VU.Vector Int -> Int -> VU.Vector Double -> VU.Vector Double+sumDblDense g nGroups v+ | nGroups <= seqFloatGroups || not (shouldPar n) =+ unsafePerformIO (sumDblChunk g v nGroups 0 n >>= VU.unsafeFreeze)+ | otherwise = unsafePerformIO $ do+ parts <- runPartialsOver n capabilities (sumDblChunk g v nGroups)+ mergeDblSum nGroups parts+ where+ !n = VU.length v+{-# NOINLINE sumDblDense #-}++sumDblChunk ::+ VU.Vector Int ->+ VU.Vector Double ->+ Int ->+ Int ->+ Int ->+ IO (VUM.IOVector Double)+sumDblChunk g v nGroups lo hi = do+ acc <- VUM.replicate nGroups (0 :: Double)+ let go !i+ | i >= hi = pure ()+ | otherwise = do+ let !k = VU.unsafeIndex g i+ c <- VUM.unsafeRead acc k+ VUM.unsafeWrite acc k (c + VU.unsafeIndex v i)+ go (i + 1)+ go lo+ pure acc++{- | Double mean: fused (Double sum, count) per group, divided once at finalize.+Same order policy as 'sumDblDense' (sequential row order is byte-identical to+the gather kernel; the chunked variant changes the float summation order).+-}+meanDblDense :: VU.Vector Int -> Int -> VU.Vector Double -> VU.Vector Double+meanDblDense g nGroups v+ | nGroups <= seqFloatGroups || not (shouldPar n) = unsafePerformIO $ do+ (s, c) <- meanDblChunk g v nGroups 0 n+ finalizeMeanDbl nGroups s c+ | otherwise = unsafePerformIO $ do+ parts <- runPartialsOver n capabilities (meanDblChunk g v nGroups)+ (s, c) <- mergeMeanDbl nGroups parts+ finalizeMeanDbl nGroups s c+ where+ !n = VU.length v+{-# NOINLINE meanDblDense #-}++meanDblChunk ::+ VU.Vector Int ->+ VU.Vector Double ->+ Int ->+ Int ->+ Int ->+ IO (VUM.IOVector Double, VUM.IOVector Int)+meanDblChunk g v nGroups lo hi = do+ s <- VUM.replicate nGroups (0 :: Double)+ c <- VUM.replicate nGroups (0 :: Int)+ let go !i+ | i >= hi = pure ()+ | otherwise = do+ let !k = VU.unsafeIndex g i+ sv <- VUM.unsafeRead s k+ VUM.unsafeWrite s k (sv + VU.unsafeIndex v i)+ cv <- VUM.unsafeRead c k+ VUM.unsafeWrite c k (cv + 1)+ go (i + 1)+ go lo+ pure (s, c)++finalizeMeanDbl ::+ Int -> VUM.IOVector Double -> VUM.IOVector Int -> IO (VU.Vector Double)+finalizeMeanDbl nGroups s c = do+ out <- VUM.new nGroups+ let go !k+ | k >= nGroups = pure ()+ | otherwise = do+ sv <- VUM.unsafeRead s k+ cv <- VUM.unsafeRead c k+ VUM.unsafeWrite+ out+ k+ (if cv == 0 then 0 / 0 else sv / fromIntegral cv)+ go (k + 1)+ go 0+ VU.unsafeFreeze out++-------------------------------------------------------------------------------+-- Integer min / max (order-independent)+-------------------------------------------------------------------------------++extremaIntDense ::+ Bool -> VU.Vector Int -> Int -> VU.Vector Int -> VU.Vector Int+extremaIntDense isMin g nGroups v = unsafePerformIO $ do+ parts <-+ runPartialsOver (VU.length v) capabilities (extremaIntChunk isMin g v nGroups)+ mergeExtremaInt isMin nGroups parts+{-# NOINLINE extremaIntDense #-}++extremaIntChunk ::+ Bool ->+ VU.Vector Int ->+ VU.Vector Int ->+ Int ->+ Int ->+ Int ->+ IO (VUM.IOVector Int)+extremaIntChunk isMin g v nGroups lo hi = do+ let !seed = if isMin then maxBound else minBound+ combine a b = if isMin then min a b else max a b+ acc <- VUM.replicate nGroups seed+ let go !i+ | i >= hi = pure ()+ | otherwise = do+ let !k = VU.unsafeIndex g i+ c <- VUM.unsafeRead acc k+ VUM.unsafeWrite acc k (combine c (VU.unsafeIndex v i))+ go (i + 1)+ go lo+ pure acc++-------------------------------------------------------------------------------+-- Double min / max (order-independent: exact per-worker merge)+-------------------------------------------------------------------------------++extremaDblDense ::+ Bool -> VU.Vector Int -> Int -> VU.Vector Double -> VU.Vector Double+extremaDblDense isMin g nGroups v+ | not (shouldPar n) =+ unsafePerformIO (extremaDblChunk isMin g v nGroups 0 n >>= VU.unsafeFreeze)+ | otherwise = unsafePerformIO $ do+ parts <- runPartialsOver n capabilities (extremaDblChunk isMin g v nGroups)+ mergeExtremaDbl isMin nGroups parts+ where+ !n = VU.length v+{-# NOINLINE extremaDblDense #-}++extremaDblChunk ::+ Bool ->+ VU.Vector Int ->+ VU.Vector Double ->+ Int ->+ Int ->+ Int ->+ IO (VUM.IOVector Double)+extremaDblChunk isMin g v nGroups lo hi = do+ let !seed = if isMin then 1 / 0 else negate (1 / 0)+ combine a b = if isMin then min a b else max a b+ acc <- VUM.replicate nGroups seed+ let go !i+ | i >= hi = pure ()+ | otherwise = do+ let !k = VU.unsafeIndex g i+ c <- VUM.unsafeRead acc k+ VUM.unsafeWrite acc k (combine c (VU.unsafeIndex v i))+ go (i + 1)+ go lo+ pure acc++-------------------------------------------------------------------------------+-- Fused max(a) - min(b) (order-independent: exact per-worker merge)+-------------------------------------------------------------------------------++{- | One streaming pass accumulating @max a@ and @min b@ together. @maxSeed@ is+the identity of @max@ (the type's least value), @minSeed@ of @min@ (its+greatest).+-}+maxMinusMinDense ::+ (VU.Unbox a, Num a, Ord a) =>+ a ->+ a ->+ VU.Vector Int ->+ Int ->+ VU.Vector a ->+ VU.Vector a ->+ VU.Vector a+{-# SPECIALIZE maxMinusMinDense ::+ Int ->+ Int ->+ VU.Vector Int ->+ Int ->+ VU.Vector Int ->+ VU.Vector Int ->+ VU.Vector Int+ #-}+{-# SPECIALIZE maxMinusMinDense ::+ Double ->+ Double ->+ VU.Vector Int ->+ Int ->+ VU.Vector Double ->+ VU.Vector Double ->+ VU.Vector Double+ #-}+maxMinusMinDense maxSeed minSeed g nGroups va vb+ | not (shouldPar n) = unsafePerformIO $ do+ (mx, mn) <- maxMinusMinChunk maxSeed minSeed g va vb nGroups 0 n+ finalizeMaxMinusMin nGroups mx mn+ | otherwise = unsafePerformIO $ do+ parts <-+ runPartialsOver+ n+ capabilities+ (maxMinusMinChunk maxSeed minSeed g va vb nGroups)+ (mx, mn) <- mergeMaxMin nGroups parts+ finalizeMaxMinusMin nGroups mx mn+ where+ !n = VU.length va+{- INLINEABLE (not NOINLINE) so the SPECIALIZE pragmas above take effect; the+kernel is a pure function of its arguments, so the usual unsafePerformIO+sharing concern does not apply. -}+{-# INLINEABLE maxMinusMinDense #-}++maxMinusMinChunk ::+ (VU.Unbox a, Ord a) =>+ a ->+ a ->+ VU.Vector Int ->+ VU.Vector a ->+ VU.Vector a ->+ Int ->+ Int ->+ Int ->+ IO (VUM.IOVector a, VUM.IOVector a)+maxMinusMinChunk maxSeed minSeed g va vb nGroups lo hi = do+ mx <- VUM.replicate nGroups maxSeed+ mn <- VUM.replicate nGroups minSeed+ let go !i+ | i >= hi = pure ()+ | otherwise = do+ let !k = VU.unsafeIndex g i+ cx <- VUM.unsafeRead mx k+ VUM.unsafeWrite mx k (max cx (VU.unsafeIndex va i))+ cn <- VUM.unsafeRead mn k+ VUM.unsafeWrite mn k (min cn (VU.unsafeIndex vb i))+ go (i + 1)+ go lo+ pure (mx, mn)++mergeMaxMin ::+ (VU.Unbox a, Ord a) =>+ Int ->+ [(VUM.IOVector a, VUM.IOVector a)] ->+ IO (VUM.IOVector a, VUM.IOVector a)+mergeMaxMin nGroups parts = case parts of+ [] -> error "mergeMaxMin: no partials"+ ((mx0, mn0) : rest) -> do+ let add (mx, mn) = do+ let go !k+ | k >= nGroups = pure ()+ | otherwise = do+ xa <- VUM.unsafeRead mx0 k+ xb <- VUM.unsafeRead mx k+ VUM.unsafeWrite mx0 k (max xa xb)+ na <- VUM.unsafeRead mn0 k+ nb <- VUM.unsafeRead mn k+ VUM.unsafeWrite mn0 k (min na nb)+ go (k + 1)+ go 0+ mapM_ add rest+ pure (mx0, mn0)++finalizeMaxMinusMin ::+ (VU.Unbox a, Num a) =>+ Int ->+ VUM.IOVector a ->+ VUM.IOVector a ->+ IO (VU.Vector a)+finalizeMaxMinusMin nGroups mx mn = do+ out <- VUM.new nGroups+ let go !k+ | k >= nGroups = pure ()+ | otherwise = do+ a <- VUM.unsafeRead mx k+ b <- VUM.unsafeRead mn k+ VUM.unsafeWrite out k (a - b)+ go (k + 1)+ go 0+ VU.unsafeFreeze out++-------------------------------------------------------------------------------+-- Top-2 sum (order-independent multiset selection; one float add at finalize)+-------------------------------------------------------------------------------++{- | Sum of the two largest values per group. Each accumulator holds the+(largest, second-largest) pair seen so far; merging two pairs keeps the top two+of the four candidates. No float ADDITION happens until the single @m1 + m2@ at+finalize, so the result is byte-identical to the gather kernel regardless of+chunking. Mirrors the gather kernel exactly, including the @realToFrac@ per+element and the @-inf -> 0@ guards at finalize.+-}+top2Dense ::+ (VU.Unbox a, Real a) => VU.Vector Int -> Int -> VU.Vector a -> VU.Vector Double+{-# SPECIALIZE top2Dense ::+ VU.Vector Int -> Int -> VU.Vector Int -> VU.Vector Double+ #-}+{-# SPECIALIZE top2Dense ::+ VU.Vector Int -> Int -> VU.Vector Double -> VU.Vector Double+ #-}+top2Dense g nGroups v+ | not (shouldPar n) = unsafePerformIO $ do+ (m1, m2) <- top2Chunk g v nGroups 0 n+ finalizeTop2 nGroups m1 m2+ | otherwise = unsafePerformIO $ do+ parts <- runPartialsOver n capabilities (top2Chunk g v nGroups)+ (m1, m2) <- mergeTop2 nGroups parts+ finalizeTop2 nGroups m1 m2+ where+ !n = VU.length v+{- INLINEABLE (not NOINLINE) so the SPECIALIZE pragmas above take effect; pure+function of its arguments, so unsafePerformIO sharing is not a concern. -}+{-# INLINEABLE top2Dense #-}++{- | Second-largest value per group: the exact same per-worker+(largest, second-largest) accumulator and merge as 'top2Dense'+('top2Chunk'/'mergeTop2'), finalized to the second max alone. A group of+size < 2 finalizes its @-inf@ seed to NaN — documented behaviour (the+db-benchmark Q8 data has no size-1 @id6@ groups). Order-independent multiset+selection, so byte-identical at any @-N@.+-}+top2SndDense ::+ (VU.Unbox a, Real a) => VU.Vector Int -> Int -> VU.Vector a -> VU.Vector Double+{-# SPECIALIZE top2SndDense ::+ VU.Vector Int -> Int -> VU.Vector Int -> VU.Vector Double+ #-}+{-# SPECIALIZE top2SndDense ::+ VU.Vector Int -> Int -> VU.Vector Double -> VU.Vector Double+ #-}+top2SndDense g nGroups v+ | not (shouldPar n) = unsafePerformIO $ do+ (m1, m2) <- top2Chunk g v nGroups 0 n+ finalizeTop2Snd nGroups m1 m2+ | otherwise = unsafePerformIO $ do+ parts <- runPartialsOver n capabilities (top2Chunk g v nGroups)+ (m1, m2) <- mergeTop2 nGroups parts+ finalizeTop2Snd nGroups m1 m2+ where+ !n = VU.length v+{- INLINEABLE (not NOINLINE) so the SPECIALIZE pragmas above take effect; pure+function of its arguments, so unsafePerformIO sharing is not a concern. -}+{-# INLINEABLE top2SndDense #-}++top2Chunk ::+ (VU.Unbox a, Real a) =>+ VU.Vector Int ->+ VU.Vector a ->+ Int ->+ Int ->+ Int ->+ IO (VUM.IOVector Double, VUM.IOVector Double)+top2Chunk g v nGroups lo hi = do+ let ninf = negate (1 / 0) :: Double+ m1 <- VUM.replicate nGroups ninf+ m2 <- VUM.replicate nGroups ninf+ let go !i+ | i >= hi = pure ()+ | otherwise = do+ let !k = VU.unsafeIndex g i+ !x = realToFrac (VU.unsafeIndex v i)+ a1 <- VUM.unsafeRead m1 k+ if x > a1+ then do+ VUM.unsafeWrite m1 k x+ VUM.unsafeWrite m2 k a1+ else do+ a2 <- VUM.unsafeRead m2 k+ when (x > a2) (VUM.unsafeWrite m2 k x)+ go (i + 1)+ go lo+ pure (m1, m2)++{- | Top two of the four candidates @{a1, a2, b1, b2}@ per group (each pair+already ordered @m1 >= m2@, @-inf@ seeds included).+-}+mergeTop2 ::+ Int ->+ [(VUM.IOVector Double, VUM.IOVector Double)] ->+ IO (VUM.IOVector Double, VUM.IOVector Double)+mergeTop2 nGroups parts = case parts of+ [] -> error "mergeTop2: no partials"+ ((m10, m20) : rest) -> do+ let add (m1, m2) = do+ let go !k+ | k >= nGroups = pure ()+ | otherwise = do+ a1 <- VUM.unsafeRead m10 k+ a2 <- VUM.unsafeRead m20 k+ b1 <- VUM.unsafeRead m1 k+ b2 <- VUM.unsafeRead m2 k+ if b1 > a1+ then do+ VUM.unsafeWrite m10 k b1+ VUM.unsafeWrite m20 k (max a1 b2)+ else VUM.unsafeWrite m20 k (max a2 b1)+ go (k + 1)+ go 0+ mapM_ add rest+ pure (m10, m20)++finalizeTop2 ::+ Int -> VUM.IOVector Double -> VUM.IOVector Double -> IO (VU.Vector Double)+finalizeTop2 nGroups m1 m2 = do+ out <- VUM.new nGroups+ let go !k+ | k >= nGroups = pure ()+ | otherwise = do+ a1 <- VUM.unsafeRead m1 k+ a2 <- VUM.unsafeRead m2 k+ let s = (if isInfinite a1 then 0 else a1) + (if isInfinite a2 then 0 else a2)+ VUM.unsafeWrite out k s+ go (k + 1)+ go 0+ VU.unsafeFreeze out++finalizeTop2Snd ::+ Int -> VUM.IOVector Double -> VUM.IOVector Double -> IO (VU.Vector Double)+finalizeTop2Snd nGroups _m1 m2 = do+ out <- VUM.new nGroups+ let go !k+ | k >= nGroups = pure ()+ | otherwise = do+ a2 <- VUM.unsafeRead m2 k+ VUM.unsafeWrite out k (if isInfinite a2 then 0 / 0 else a2)+ go (k + 1)+ go 0+ VU.unsafeFreeze out++-------------------------------------------------------------------------------+-- Integer mean (exact integer sum + count, divided once -> order-independent)+-------------------------------------------------------------------------------++{- | Integer mean in ONE fused pass: a running integer sum and count per group,+divided once at finalize. The integer sum is exact, so the parallel partial+merge is byte-identical to the sequential single pass at any @-N@.+-}+meanIntDense :: VU.Vector Int -> Int -> VU.Vector Int -> VU.Vector Double+meanIntDense g nGroups v+ | not (shouldPar n) = unsafePerformIO $ do+ (s, c) <- meanIntChunk g v nGroups 0 n+ finalizeMeanInt nGroups s c+ | otherwise = unsafePerformIO $ do+ parts <- runPartialsOver n capabilities (meanIntChunk g v nGroups)+ (s, c) <- mergePair nGroups parts+ finalizeMeanInt nGroups s c+ where+ !n = VU.length v+{-# NOINLINE meanIntDense #-}++meanIntChunk ::+ VU.Vector Int ->+ VU.Vector Int ->+ Int ->+ Int ->+ Int ->+ IO (VUM.IOVector Int, VUM.IOVector Int)+meanIntChunk g v nGroups lo hi = do+ s <- VUM.replicate nGroups (0 :: Int)+ c <- VUM.replicate nGroups (0 :: Int)+ let go !i+ | i >= hi = pure ()+ | otherwise = do+ let !k = VU.unsafeIndex g i+ sv <- VUM.unsafeRead s k+ VUM.unsafeWrite s k (sv + VU.unsafeIndex v i)+ cv <- VUM.unsafeRead c k+ VUM.unsafeWrite c k (cv + 1)+ go (i + 1)+ go lo+ pure (s, c)++finalizeMeanInt ::+ Int -> VUM.IOVector Int -> VUM.IOVector Int -> IO (VU.Vector Double)+finalizeMeanInt nGroups s c = do+ out <- VUM.new nGroups+ let go !k+ | k >= nGroups = pure ()+ | otherwise = do+ sv <- VUM.unsafeRead s k+ cv <- VUM.unsafeRead c k+ VUM.unsafeWrite+ out+ k+ (if cv == 0 then 0 / 0 else fromIntegral sv / fromIntegral cv)+ go (k + 1)+ go 0+ VU.unsafeFreeze out++-------------------------------------------------------------------------------+-- Partial accumulation + merge+-------------------------------------------------------------------------------++mergeIntSum :: Int -> [VUM.IOVector Int] -> IO (VU.Vector Int)+mergeIntSum nGroups parts = case parts of+ [] -> VU.unsafeFreeze =<< VUM.replicate nGroups 0+ (p0 : rest) -> do+ let add !p = do+ let go !k+ | k >= nGroups = pure ()+ | otherwise = do+ a <- VUM.unsafeRead p0 k+ b <- VUM.unsafeRead p k+ VUM.unsafeWrite p0 k (a + b)+ go (k + 1)+ go 0+ mapM_ add rest+ VU.unsafeFreeze p0++{- | Sum the per-worker Double partials into the first worker's accumulator IN+WORKER ORDER: deterministic at a fixed @-N@, but the float summation order is+chunk-major rather than the sequential row order.+-}+mergeDblSum :: Int -> [VUM.IOVector Double] -> IO (VU.Vector Double)+mergeDblSum nGroups parts = case parts of+ [] -> VU.unsafeFreeze =<< VUM.replicate nGroups 0+ (p0 : rest) -> do+ let add !p = do+ let go !k+ | k >= nGroups = pure ()+ | otherwise = do+ a <- VUM.unsafeRead p0 k+ b <- VUM.unsafeRead p k+ VUM.unsafeWrite p0 k (a + b)+ go (k + 1)+ go 0+ mapM_ add rest+ VU.unsafeFreeze p0++{- | Merge per-worker (sum, count) partials into the first worker's pair by+exact integer addition; returns the accumulated pair for finalize.+-}+mergePair ::+ Int ->+ [(VUM.IOVector Int, VUM.IOVector Int)] ->+ IO (VUM.IOVector Int, VUM.IOVector Int)+mergePair nGroups parts = case parts of+ [] -> (,) <$> VUM.replicate nGroups 0 <*> VUM.replicate nGroups 0+ ((s0, c0) : rest) -> do+ let add (s, c) = do+ let go !k+ | k >= nGroups = pure ()+ | otherwise = do+ sa <- VUM.unsafeRead s0 k+ sb <- VUM.unsafeRead s k+ VUM.unsafeWrite s0 k (sa + sb)+ ca <- VUM.unsafeRead c0 k+ cb <- VUM.unsafeRead c k+ VUM.unsafeWrite c0 k (ca + cb)+ go (k + 1)+ go 0+ mapM_ add rest+ pure (s0, c0)++{- | As 'mergePair' but for the Double (sum, count) partials of the Double mean;+worker-order float sums (see 'mergeDblSum').+-}+mergeMeanDbl ::+ Int ->+ [(VUM.IOVector Double, VUM.IOVector Int)] ->+ IO (VUM.IOVector Double, VUM.IOVector Int)+mergeMeanDbl nGroups parts = case parts of+ [] -> (,) <$> VUM.replicate nGroups 0 <*> VUM.replicate nGroups 0+ ((s0, c0) : rest) -> do+ let add (s, c) = do+ let go !k+ | k >= nGroups = pure ()+ | otherwise = do+ sa <- VUM.unsafeRead s0 k+ sb <- VUM.unsafeRead s k+ VUM.unsafeWrite s0 k (sa + sb)+ ca <- VUM.unsafeRead c0 k+ cb <- VUM.unsafeRead c k+ VUM.unsafeWrite c0 k (ca + cb)+ go (k + 1)+ go 0+ mapM_ add rest+ pure (s0, c0)++mergeExtremaInt :: Bool -> Int -> [VUM.IOVector Int] -> IO (VU.Vector Int)+mergeExtremaInt isMin nGroups parts = case parts of+ [] ->+ VU.unsafeFreeze =<< VUM.replicate nGroups (if isMin then maxBound else minBound)+ (p0 : rest) -> do+ let combine a b = if isMin then min a b else max a b+ add !p = do+ let go !k+ | k >= nGroups = pure ()+ | otherwise = do+ a <- VUM.unsafeRead p0 k+ b <- VUM.unsafeRead p k+ VUM.unsafeWrite p0 k (combine a b)+ go (k + 1)+ go 0+ mapM_ add rest+ VU.unsafeFreeze p0++mergeExtremaDbl :: Bool -> Int -> [VUM.IOVector Double] -> IO (VU.Vector Double)+mergeExtremaDbl isMin nGroups parts = case parts of+ [] ->+ VU.unsafeFreeze+ =<< VUM.replicate nGroups (if isMin then 1 / 0 else negate (1 / 0))+ (p0 : rest) -> do+ let combine a b = if isMin then min a b else max a b+ add !p = do+ let go !k+ | k >= nGroups = pure ()+ | otherwise = do+ a <- VUM.unsafeRead p0 k+ b <- VUM.unsafeRead p k+ VUM.unsafeWrite p0 k (combine a b)+ go (k + 1)+ go 0+ mapM_ add rest+ VU.unsafeFreeze p0
+ src-internal/DataFrame/Internal/Aggregation/Kernel/Fused.hs view
@@ -0,0 +1,757 @@+{-# LANGUAGE BangPatterns #-}+{-# LANGUAGE ExplicitNamespaces #-}+{-# LANGUAGE GADTs #-}+{-# LANGUAGE ScopedTypeVariables #-}+{-# LANGUAGE TypeApplications #-}++{- | Fused multi-reduction aggregation kernels: several reductions over the+SAME grouping, evaluated in one pass over the rows.++Two shapes, picked by group-domain width:++* 'mkFusedAgg' \/ 'runFusedAggs' — the STREAMING pass, driven by @rowToGroup@.+ No @valueIndices@ and no placement pass; each worker keeps a private+ accumulator per reduction and the partials merge over group slices. Capped at+ 'streamGroupCap', above which the per-worker arrays overflow cache.++* 'mkGatherAgg' \/ 'runGatherAggs' — the GATHER pass, driven by the grouped+ @(valueIndices, offsets)@ layout, for group domains too wide to stream.++Both amortize the memory traffic of the grouping across every reduction in the+batch instead of re-reading it once per reduction.+-}+module DataFrame.Internal.Aggregation.Kernel.Fused (+ FusedAgg,+ mkFusedAgg,+ runFusedAggs,+ GatherAgg,+ mkGatherAgg,+ runGatherAggs,+) where++import Control.Exception (evaluate)+import Control.Monad (replicateM, when)+import Data.Type.Equality (TestEquality (..), type (:~:) (Refl))+import qualified Data.Vector.Unboxed as VU+import qualified Data.Vector.Unboxed.Mutable as VUM+import System.IO.Unsafe (unsafePerformIO)+import Type.Reflection (typeRep)++import DataFrame.Internal.Aggregation.Kernel.Scatter (+ groupRangeBounds,+ groupSlices,+ overGroupsAcc,+ streamGroupCap,+ )+import DataFrame.Internal.Aggregation.Reduction (Reduction (..))+import DataFrame.Internal.Column (Column (..), fromUnboxedVector)+import DataFrame.Internal.Control.Concurrent (+ capabilities,+ forkJoin,+ parThreshold,+ parallelBounds_,+ shouldParallelize,+ )++-- | Whether to fan out at this row count.+shouldPar :: Int -> Bool+shouldPar = shouldParallelize parThreshold++-------------------------------------------------------------------------------+-- Fused multi-reduction streaming pass+-------------------------------------------------------------------------------++{- | One fused reduction of a multi-reduction streaming pass: how to allocate a+per-worker accumulator, fold a row range into it, merge another worker's+accumulator into it over a group range (callers merge in worker order), and+finalize the fully merged accumulator into the output column.+-}+data FusedAgg+ = forall s.+ FusedAgg+ (IO s)+ -- \^ allocate one worker's accumulator+ (s -> Int -> Int -> IO ())+ -- \^ accumulate rows [lo, hi) in ascending order+ (s -> s -> Int -> Int -> IO ())+ -- \^ merge the second accumulator into the first over groups [lo, hi)+ (s -> IO Column)++-- \^ finalize the merged accumulator++{- | Below this many groups a Double sum/mean does NOT join the fused streaming+pass: the per-expression kernel it would replace+('DataFrame.Internal.AggKernelDirect.sumDblDirect' under its @seqFloatGroups@+policy, value mirrored here) runs those as ONE sequential row-order pass that+is byte-identical to the gather kernels and the interpreter, and the vectorized+parity gate asserts exactly that. Above it the per-expression kernel already+merges chunk partials in worker order, so fusing changes nothing semantically+new. Int reductions and count/min/max are exact under any merge and always+fuse.+-}+fusedSeqFloatGroups :: Int+fusedSeqFloatGroups = 65536++{- | Build the fused-pass reduction for one @(reduction, column)@ pair, or+'Nothing' when the pair cannot stream (nullable/boxed columns; the+order-sensitive var/std/top2 reductions, which keep their per-expression+kernels; or a Double sum/mean below 'fusedSeqFloatGroups', which keeps its+byte-identical sequential pass). Sum/min/max/count/mean over Int are exact+under any chunk merge; the admitted Double sum and mean merge their per-worker+partials in worker order (deterministic at a fixed @-N@, chunk-major float+summation order).+-}+mkFusedAgg :: Int -> VU.Vector Int -> Reduction -> Column -> Maybe FusedAgg+mkFusedAgg nGroups rtg red col+ | nGroups <= 0 || nGroups > streamGroupCap = Nothing+ | otherwise = case col of+ UnboxedColumn Nothing (v :: VU.Vector a)+ | Just Refl <- testEquality (typeRep @a) (typeRep @Int) ->+ case red of+ RCount -> Just (countFusedAgg nGroups rtg)+ RSum -> Just (sumIntFusedAgg nGroups rtg v)+ RMean -> Just (meanIntFusedAgg nGroups rtg v)+ RMin -> Just (extremaIntFusedAgg True nGroups rtg v)+ RMax -> Just (extremaIntFusedAgg False nGroups rtg v)+ _ -> Nothing+ | Just Refl <- testEquality (typeRep @a) (typeRep @Double) ->+ case red of+ RCount -> Just (countFusedAgg nGroups rtg)+ RSum+ | nGroups > fusedSeqFloatGroups ->+ Just (sumDblFusedAgg nGroups rtg v)+ RMean+ | nGroups > fusedSeqFloatGroups ->+ Just (meanDblFusedAgg nGroups rtg v)+ RMin -> Just (extremaDblFusedAgg True nGroups rtg v)+ RMax -> Just (extremaDblFusedAgg False nGroups rtg v)+ -- Top-2 selection is an exact multiset selection (no float+ -- arithmetic before finalize), so its per-worker merge is+ -- byte-identical to the per-expression kernels at any -N.+ RTop2Snd -> Just (top2SndDblFusedAgg nGroups rtg v)+ _ -> Nothing+ _ -> Nothing++{- | Execute all fused reductions in ONE pass over @rowToGroup@ and the value+columns: the rows are split into one contiguous chunk per capability, each+worker walks its chunk in 4096-row blocks running every reduction's+accumulate-step on the block (the block's @rowToGroup@ slice stays in L1+across the k steps), and each reduction then merges its per-worker partials in+fixed worker order (parallel over group slices) and finalizes. Per-group update+order within each worker is ascending original row order, and chunk boundaries+are a fixed function of the row and capability counts, so the result is+deterministic at a fixed @-N@; on a single capability it is bit-identical to+running each unfused sequential scatter kernel separately.++Pure w.r.t. its immutable inputs (deterministic fan-out and merge order), so+the 'unsafePerformIO' is safe.+-}+runFusedAggs :: Int -> Int -> [FusedAgg] -> [Column]+runFusedAggs n nGroups aggs = unsafePerformIO $ do+ let !caps' = if shouldPar n then capabilities else 1+ !per = (max 1 n + caps' - 1) `div` caps'+ opened <- mapM (openFusedAgg caps' nGroups) aggs++ let stepsFor w = map (\(steps, _) -> steps !! w) opened+ _ <-+ forkJoin+ [ blockRun (stepsFor w) lo hi+ | w <- [0 .. caps' - 1]+ , let lo = min n (w * per)+ , let hi = min n (lo + per)+ ]+ mapM snd opened+{-# NOINLINE runFusedAggs #-}++{- | Open one fused reduction for @caps'@ workers: its per-worker step+functions (worker order) and the merge+finalize action.+-}+openFusedAgg :: Int -> Int -> FusedAgg -> IO ([Int -> Int -> IO ()], IO Column)+openFusedAgg caps' nGroups (FusedAgg new step mergeR fin) = do+ ss <- replicateM caps' new+ let finish = case ss of+ [] -> error "runFusedAggs: no workers"+ (s0 : rest) -> do+ _ <-+ forkJoin+ [ mapM_ (\s -> mergeR s0 s lo hi) rest+ | (lo, hi) <- groupSlices nGroups+ ]+ fin s0+ pure (map step ss, finish)++-- | Rows per fused block: the block's rowToGroup slice (32KB) stays in L1.+fusedBlock :: Int+fusedBlock = 4096++blockRun :: [Int -> Int -> IO ()] -> Int -> Int -> IO ()+blockRun steps lo0 hi = go lo0+ where+ go !lo+ | lo >= hi = pure ()+ | otherwise = do+ let !e = min hi (lo + fusedBlock)+ mapM_ (\s -> s lo e) steps+ go e++-- Individual fused reductions. Each step loop is monomorphic.++countFusedAgg :: Int -> VU.Vector Int -> FusedAgg+countFusedAgg nGroups rtg =+ FusedAgg+ (VUM.replicate nGroups (0 :: Int))+ (countStepK rtg)+ addIntRange+ (fmap fromUnboxedVector . VU.unsafeFreeze)++countStepK :: VU.Vector Int -> VUM.IOVector Int -> Int -> Int -> IO ()+countStepK rtg acc lo hi = go lo+ where+ go !i+ | i >= hi = pure ()+ | otherwise = do+ let !k = VU.unsafeIndex rtg i+ c <- VUM.unsafeRead acc k+ VUM.unsafeWrite acc k (c + 1)+ go (i + 1)++sumIntFusedAgg :: Int -> VU.Vector Int -> VU.Vector Int -> FusedAgg+sumIntFusedAgg nGroups rtg v =+ FusedAgg+ (VUM.replicate nGroups (0 :: Int))+ (sumStepInt rtg v)+ addIntRange+ (fmap fromUnboxedVector . VU.unsafeFreeze)++sumStepInt ::+ VU.Vector Int -> VU.Vector Int -> VUM.IOVector Int -> Int -> Int -> IO ()+sumStepInt rtg v acc lo hi = go lo+ where+ go !i+ | i >= hi = pure ()+ | otherwise = do+ let !k = VU.unsafeIndex rtg i+ c <- VUM.unsafeRead acc k+ VUM.unsafeWrite acc k (c + VU.unsafeIndex v i)+ go (i + 1)++sumDblFusedAgg :: Int -> VU.Vector Int -> VU.Vector Double -> FusedAgg+sumDblFusedAgg nGroups rtg v =+ FusedAgg+ (VUM.replicate nGroups (0 :: Double))+ (sumStepDbl rtg v)+ addDblRange+ (fmap fromUnboxedVector . VU.unsafeFreeze)++sumStepDbl ::+ VU.Vector Int -> VU.Vector Double -> VUM.IOVector Double -> Int -> Int -> IO ()+sumStepDbl rtg v acc lo hi = go lo+ where+ go !i+ | i >= hi = pure ()+ | otherwise = do+ let !k = VU.unsafeIndex rtg i+ c <- VUM.unsafeRead acc k+ VUM.unsafeWrite acc k (c + VU.unsafeIndex v i)+ go (i + 1)++{- | The mean aggs hold their (sum, count) pair INTERLEAVED in one array —+slots @2g@/@2g+1@ share a cache line (pairs are 16-byte aligned, so they never+straddle one), halving the accumulator misses of the random per-row update+against two separate arrays (measured ~13% off a fused sum+mean pass at 1e6+groups / 1e8 rows on -N16). The count is exact in both layouts (an integer, or+integer-valued Double additions well below 2^53), so sums, merges and the+finalize divide are bit-identical to the two-array layout.+-}+meanIntFusedAgg :: Int -> VU.Vector Int -> VU.Vector Int -> FusedAgg+meanIntFusedAgg nGroups rtg v =+ FusedAgg+ (VUM.replicate (2 * nGroups) (0 :: Int))+ (meanStepInt rtg v)+ (\a b lo hi -> addIntRange a b (2 * lo) (2 * hi))+ ( \s -> do+ sv <- VU.unsafeFreeze s+ pure+ ( fromUnboxedVector+ ( VU.generate+ nGroups+ ( \g ->+ let !sx = VU.unsafeIndex sv (2 * g)+ !cx = VU.unsafeIndex sv (2 * g + 1)+ in if cx == 0+ then 0 / 0+ else fromIntegral sx / fromIntegral cx :: Double+ )+ )+ )+ )++meanStepInt ::+ VU.Vector Int ->+ VU.Vector Int ->+ VUM.IOVector Int ->+ Int ->+ Int ->+ IO ()+meanStepInt rtg v s lo hi = go lo+ where+ go !i+ | i >= hi = pure ()+ | otherwise = do+ let !k2 = 2 * VU.unsafeIndex rtg i+ sv <- VUM.unsafeRead s k2+ VUM.unsafeWrite s k2 (sv + VU.unsafeIndex v i)+ cv <- VUM.unsafeRead s (k2 + 1)+ VUM.unsafeWrite s (k2 + 1) (cv + 1)+ go (i + 1)++-- | See 'meanIntFusedAgg' for the interleaved accumulator layout.+meanDblFusedAgg :: Int -> VU.Vector Int -> VU.Vector Double -> FusedAgg+meanDblFusedAgg nGroups rtg v =+ FusedAgg+ (VUM.replicate (2 * nGroups) (0 :: Double))+ (meanStepDbl rtg v)+ (\a b lo hi -> addDblRange a b (2 * lo) (2 * hi))+ ( \s -> do+ sv <- VU.unsafeFreeze s+ pure+ ( fromUnboxedVector+ ( VU.generate+ nGroups+ ( \g ->+ let !sx = VU.unsafeIndex sv (2 * g)+ !cx = VU.unsafeIndex sv (2 * g + 1)+ in if cx == 0 then 0 / 0 else sx / cx+ )+ )+ )+ )++meanStepDbl ::+ VU.Vector Int ->+ VU.Vector Double ->+ VUM.IOVector Double ->+ Int ->+ Int ->+ IO ()+meanStepDbl rtg v s lo hi = go lo+ where+ go !i+ | i >= hi = pure ()+ | otherwise = do+ let !k2 = 2 * VU.unsafeIndex rtg i+ sv <- VUM.unsafeRead s k2+ VUM.unsafeWrite s k2 (sv + VU.unsafeIndex v i)+ cv <- VUM.unsafeRead s (k2 + 1)+ VUM.unsafeWrite s (k2 + 1) (cv + 1)+ go (i + 1)++extremaIntFusedAgg :: Bool -> Int -> VU.Vector Int -> VU.Vector Int -> FusedAgg+extremaIntFusedAgg isMin nGroups rtg v =+ FusedAgg+ (VUM.replicate nGroups (if isMin then maxBound else minBound :: Int))+ (extremaStepInt isMin rtg v)+ (combineIntRange isMin)+ (fmap fromUnboxedVector . VU.unsafeFreeze)++extremaStepInt ::+ Bool ->+ VU.Vector Int ->+ VU.Vector Int ->+ VUM.IOVector Int ->+ Int ->+ Int ->+ IO ()+extremaStepInt isMin rtg v acc lo hi = go lo+ where+ go !i+ | i >= hi = pure ()+ | otherwise = do+ let !k = VU.unsafeIndex rtg i+ !x = VU.unsafeIndex v i+ c <- VUM.unsafeRead acc k+ VUM.unsafeWrite acc k (if isMin then min c x else max c x)+ go (i + 1)++extremaDblFusedAgg ::+ Bool -> Int -> VU.Vector Int -> VU.Vector Double -> FusedAgg+extremaDblFusedAgg isMin nGroups rtg v =+ FusedAgg+ (VUM.replicate nGroups (if isMin then 1 / 0 else negate (1 / 0) :: Double))+ (extremaStepDbl isMin rtg v)+ (combineDblRange isMin)+ (fmap fromUnboxedVector . VU.unsafeFreeze)++extremaStepDbl ::+ Bool ->+ VU.Vector Int ->+ VU.Vector Double ->+ VUM.IOVector Double ->+ Int ->+ Int ->+ IO ()+extremaStepDbl isMin rtg v acc lo hi = go lo+ where+ go !i+ | i >= hi = pure ()+ | otherwise = do+ let !k = VU.unsafeIndex rtg i+ !x = VU.unsafeIndex v i+ c <- VUM.unsafeRead acc k+ VUM.unsafeWrite acc k (if isMin then min c x else max c x)+ go (i + 1)++{- | Fused second-largest over a Double column. The per-group+(largest, second-largest) pair is INTERLEAVED at slots @2g@/@2g+1@ (one cache+line per group, as 'meanDblFusedAgg'); the update is the same top-2 selection+as every other top2 kernel, the merge keeps the top two of the four candidates+per group (exact — no float arithmetic), and the finalize returns the second+max, NaN for a group of size < 2 (the @-inf@ seed; see+'DataFrame.Internal.AggKernel.top2SndScatter').+-}+top2SndDblFusedAgg :: Int -> VU.Vector Int -> VU.Vector Double -> FusedAgg+top2SndDblFusedAgg nGroups rtg v =+ FusedAgg+ (VUM.replicate (2 * nGroups) (negate (1 / 0) :: Double))+ (top2SndStepDbl rtg v)+ mergeTop2Range+ ( \s -> do+ sv <- VU.unsafeFreeze s+ pure+ ( fromUnboxedVector+ ( VU.generate+ nGroups+ ( \g ->+ let !a2 = VU.unsafeIndex sv (2 * g + 1)+ in if isInfinite a2 then 0 / 0 else a2+ )+ )+ )+ )++top2SndStepDbl ::+ VU.Vector Int ->+ VU.Vector Double ->+ VUM.IOVector Double ->+ Int ->+ Int ->+ IO ()+top2SndStepDbl rtg v s lo hi = go lo+ where+ go !i+ | i >= hi = pure ()+ | otherwise = do+ let !k2 = 2 * VU.unsafeIndex rtg i+ !x = VU.unsafeIndex v i+ a1 <- VUM.unsafeRead s k2+ if x > a1+ then do+ VUM.unsafeWrite s k2 x+ VUM.unsafeWrite s (k2 + 1) a1+ else do+ a2 <- VUM.unsafeRead s (k2 + 1)+ when (x > a2) $ VUM.unsafeWrite s (k2 + 1) x+ go (i + 1)++-- | Top two of the four candidates per group (pairs already ordered m1 >= m2).+mergeTop2Range ::+ VUM.IOVector Double -> VUM.IOVector Double -> Int -> Int -> IO ()+mergeTop2Range a b lo hi = go lo+ where+ go !g+ | g >= hi = pure ()+ | otherwise = do+ let !g2 = 2 * g+ a1 <- VUM.unsafeRead a g2+ a2 <- VUM.unsafeRead a (g2 + 1)+ b1 <- VUM.unsafeRead b g2+ b2 <- VUM.unsafeRead b (g2 + 1)+ if b1 > a1+ then do+ VUM.unsafeWrite a g2 b1+ VUM.unsafeWrite a (g2 + 1) (max a1 b2)+ else VUM.unsafeWrite a (g2 + 1) (max a2 b1)+ go (g + 1)++addIntRange :: VUM.IOVector Int -> VUM.IOVector Int -> Int -> Int -> IO ()+addIntRange a b lo hi = go lo+ where+ go !g+ | g >= hi = pure ()+ | otherwise = do+ x <- VUM.unsafeRead a g+ y <- VUM.unsafeRead b g+ VUM.unsafeWrite a g (x + y)+ go (g + 1)++addDblRange :: VUM.IOVector Double -> VUM.IOVector Double -> Int -> Int -> IO ()+addDblRange a b lo hi = go lo+ where+ go !g+ | g >= hi = pure ()+ | otherwise = do+ x <- VUM.unsafeRead a g+ y <- VUM.unsafeRead b g+ VUM.unsafeWrite a g (x + y)+ go (g + 1)++combineIntRange ::+ Bool -> VUM.IOVector Int -> VUM.IOVector Int -> Int -> Int -> IO ()+combineIntRange isMin a b lo hi = go lo+ where+ go !g+ | g >= hi = pure ()+ | otherwise = do+ x <- VUM.unsafeRead a g+ y <- VUM.unsafeRead b g+ VUM.unsafeWrite a g (if isMin then min x y else max x y)+ go (g + 1)++combineDblRange ::+ Bool -> VUM.IOVector Double -> VUM.IOVector Double -> Int -> Int -> IO ()+combineDblRange isMin a b lo hi = go lo+ where+ go !g+ | g >= hi = pure ()+ | otherwise = do+ x <- VUM.unsafeRead a g+ y <- VUM.unsafeRead b g+ VUM.unsafeWrite a g (if isMin then min x y else max x y)+ go (g + 1)++-------------------------------------------------------------------------------+-- Fused multi-reduction gather pass (nGroups above 'streamGroupCap')+-------------------------------------------------------------------------------++{- | One fused reduction of a multi-reduction GATHER pass: allocate the output+array, fold a contiguous group range (each group's rows via the shared+@valueIndices@ slice, accumulator in registers, one write per group), finalize.+Group ranges are disjoint across workers, so there is no merge and every+reduction reproduces the exact per-group fold order and formula of its unfused+gather kernel ('reduceParTyped') — results are bit-identical to running the+kernels separately, at any @-N@.+-}+data GatherAgg+ = forall s.+ GatherAgg+ (IO s)+ (s -> Int -> Int -> IO ())+ (s -> IO Column)++{- | Build the fused gather reduction for one @(reduction, column)@ pair, or+'Nothing' when the pair cannot fuse (nullable/boxed columns, or var/std/top2,+which keep their per-expression gather kernels). @vis@/@offs@ are captured+lazily: nothing is forced until the pass actually runs.+-}+mkGatherAgg ::+ Int ->+ VU.Vector Int ->+ VU.Vector Int ->+ Reduction ->+ Column ->+ Maybe GatherAgg+mkGatherAgg nGroups vis offs red col = case col of+ UnboxedColumn Nothing (v :: VU.Vector a)+ | Just Refl <- testEquality (typeRep @a) (typeRep @Int) ->+ case red of+ RCount -> Just countGather+ RSum -> Just (outGather (gatherSumInt vis offs v))+ RMean -> Just (outGatherD (gatherMeanInt vis offs v))+ RMin -> Just (outGather (gatherExtremaInt True vis offs v))+ RMax -> Just (outGather (gatherExtremaInt False vis offs v))+ _ -> Nothing+ | Just Refl <- testEquality (typeRep @a) (typeRep @Double) ->+ case red of+ RCount -> Just countGather+ RSum -> Just (outGatherD (gatherSumDbl vis offs v))+ RMean -> Just (outGatherD (gatherMeanDbl vis offs v))+ RMin -> Just (outGatherD (gatherExtremaDbl True vis offs v))+ RMax -> Just (outGatherD (gatherExtremaDbl False vis offs v))+ _ -> Nothing+ _ -> Nothing+ where+ countGather =+ GatherAgg+ (VUM.new nGroups)+ (gatherCount offs)+ (fmap fromUnboxedVector . VU.unsafeFreeze)+ outGather step =+ GatherAgg+ (VUM.new nGroups :: IO (VUM.IOVector Int))+ step+ (fmap fromUnboxedVector . VU.unsafeFreeze)+ outGatherD step =+ GatherAgg+ (VUM.new nGroups :: IO (VUM.IOVector Double))+ step+ (fmap fromUnboxedVector . VU.unsafeFreeze)++{- | Number of groups each fused-gather block hands to every reduction before+moving on: the block's @valueIndices@ slice stays hot in cache across the k+per-reduction loops.+-}+gatherBlock :: Int+gatherBlock = 32++{- | Execute all fused gather reductions in one traversal: workers own disjoint+contiguous group ranges (row-balanced, same policy as every gather kernel), and+walk them in 'gatherBlock'-group blocks running each reduction's fold on the+block. Deterministic and bit-identical to the unfused kernels (see+'GatherAgg'). Forces @valueIndices@ once, before the fan-out.++Pure w.r.t. its immutable inputs, so the 'unsafePerformIO' is safe.+-}+runGatherAggs ::+ VU.Vector Int -> VU.Vector Int -> Int -> [GatherAgg] -> [Column]+runGatherAggs vis offs nGroups aggs = unsafePerformIO $ do+ _ <- evaluate (VU.length vis)+ let !caps = capabilities+ !bounds = groupRangeBounds offs nGroups caps+ opened <-+ mapM (\(GatherAgg new step fin) -> do s <- new; pure (step s, fin s)) aggs+ parallelBounds_ caps bounds $ \gs ge ->+ let go !g+ | g >= ge = pure ()+ | otherwise = do+ let !e = min ge (g + gatherBlock)+ mapM_ (\(st, _) -> st g e) opened+ go e+ in go gs+ mapM snd opened+{-# NOINLINE runGatherAggs #-}++-- Monomorphic per-reduction gather folds; each replicates the exact per-group+-- recurrence of its unfused kernel above ('countPar'/'sumPar'/'extremaPar'/+-- 'meanPar'), so fused results are bit-identical.++gatherCount :: VU.Vector Int -> VUM.IOVector Int -> Int -> Int -> IO ()+gatherCount offs out gs ge = go gs+ where+ go !g+ | g >= ge = pure ()+ | otherwise = do+ let !c = VU.unsafeIndex offs (g + 1) - VU.unsafeIndex offs g+ VUM.unsafeWrite out g c+ go (g + 1)++gatherSumInt ::+ VU.Vector Int ->+ VU.Vector Int ->+ VU.Vector Int ->+ VUM.IOVector Int ->+ Int ->+ Int ->+ IO ()+gatherSumInt vis offs v out gs ge =+ overGroupsAcc vis offs gs ge 0 (\acc row -> acc + VU.unsafeIndex v row) $+ VUM.unsafeWrite out++gatherSumDbl ::+ VU.Vector Int ->+ VU.Vector Int ->+ VU.Vector Double ->+ VUM.IOVector Double ->+ Int ->+ Int ->+ IO ()+gatherSumDbl vis offs v out gs ge =+ overGroupsAcc vis offs gs ge 0 (\acc row -> acc + VU.unsafeIndex v row) $+ VUM.unsafeWrite out++gatherExtremaInt ::+ Bool ->+ VU.Vector Int ->+ VU.Vector Int ->+ VU.Vector Int ->+ VUM.IOVector Int ->+ Int ->+ Int ->+ IO ()+gatherExtremaInt isMin vis offs v out gs ge =+ overGroupsAcc+ vis+ offs+ gs+ ge+ (if isMin then maxBound else minBound)+ ( \acc row ->+ let !x = VU.unsafeIndex v row+ in if isMin then min acc x else max acc x+ )+ (VUM.unsafeWrite out)++gatherExtremaDbl ::+ Bool ->+ VU.Vector Int ->+ VU.Vector Int ->+ VU.Vector Double ->+ VUM.IOVector Double ->+ Int ->+ Int ->+ IO ()+gatherExtremaDbl isMin vis offs v out gs ge =+ overGroupsAcc+ vis+ offs+ gs+ ge+ (if isMin then 1 / 0 else negate (1 / 0))+ ( \acc row ->+ let !x = VU.unsafeIndex v row+ in if isMin then min acc x else max acc x+ )+ (VUM.unsafeWrite out)++-- | Exact replica of 'meanPar''s per-group loop (Int element type).+gatherMeanInt ::+ VU.Vector Int ->+ VU.Vector Int ->+ VU.Vector Int ->+ VUM.IOVector Double ->+ Int ->+ Int ->+ IO ()+gatherMeanInt vis offs v out gs ge = grp gs+ where+ grp !g+ | g >= ge = pure ()+ | otherwise = do+ let !e = VU.unsafeIndex offs (g + 1)+ inner !pos !acc+ | pos >= e = acc+ | otherwise =+ inner+ (pos + 1)+ (acc + fromIntegral (VU.unsafeIndex v (VU.unsafeIndex vis pos)))+ !s0 = VU.unsafeIndex offs g+ !total = inner s0 0+ !c = e - s0+ VUM.unsafeWrite out g (if c == 0 then 0 / 0 else total / fromIntegral c)+ grp (g + 1)++-- | Exact replica of 'meanPar''s per-group loop (Double element type).+gatherMeanDbl ::+ VU.Vector Int ->+ VU.Vector Int ->+ VU.Vector Double ->+ VUM.IOVector Double ->+ Int ->+ Int ->+ IO ()+gatherMeanDbl vis offs v out gs ge = grp gs+ where+ grp !g+ | g >= ge = pure ()+ | otherwise = do+ let !e = VU.unsafeIndex offs (g + 1)+ inner !pos !acc+ | pos >= e = acc+ | otherwise =+ inner+ (pos + 1)+ (acc + VU.unsafeIndex v (VU.unsafeIndex vis pos))+ !s0 = VU.unsafeIndex offs g+ !total = inner s0 0+ !c = e - s0+ VUM.unsafeWrite out g (if c == 0 then 0 / 0 else total / fromIntegral c)+ grp (g + 1)
+ src-internal/DataFrame/Internal/Aggregation/Kernel/Moments.hs view
@@ -0,0 +1,389 @@+{-# LANGUAGE BangPatterns #-}+{-# LANGUAGE ExplicitNamespaces #-}+{-# LANGUAGE GADTs #-}+{-# LANGUAGE RankNTypes #-}+{-# LANGUAGE ScopedTypeVariables #-}+{-# LANGUAGE TypeApplications #-}++{- | The fused two-column moment kernel: one pass over @x@ and @y@ producing the+six sufficient statistics @(n, Sx, Sy, Sxx, Syy, Sxy)@ per group, from which the+whole correlation\/regression family (mean, variance, covariance, correlation,+OLS slope) is algebra requiring no further look at the rows.++The sequential and parallel passes live together because they must agree on+floating-point accumulation order. Moments are additive, so a row-range split+with a merge would be correct in exact arithmetic — but float addition is not+associative, so 'momentScatterPar' partitions by GROUP range instead: every+group accumulates start-to-finish inside a single worker, in the same order as+'momentScatter'. That is what makes the two byte-identical at any @-N@.+-}+module DataFrame.Internal.Aggregation.Kernel.Moments (+ Moments (..),+ momentScatter,+ momentScatterPar,+ momentStreamPar,+) where++import Control.Monad.ST (runST)+import Data.Type.Equality (TestEquality (..), type (:~:) (Refl))+import qualified Data.Vector.Unboxed as VU+import qualified Data.Vector.Unboxed.Mutable as VUM+import System.IO.Unsafe (unsafePerformIO)+import Type.Reflection (typeRep)++import DataFrame.Internal.Aggregation.Kernel.Scatter (+ groupRangeBounds,+ groupSlices,+ rtgFromVis,+ streamGroupCap,+ )+import DataFrame.Internal.Aggregation.Reduction (cleanDoubleVector)+import DataFrame.Internal.Column (Column (..), fromUnboxedVector)+import DataFrame.Internal.Control.Concurrent (+ capabilities,+ forkJoin,+ parThreshold,+ parallelBounds_,+ shouldParallelize,+ )++-- | Whether to fan out at this row count.+shouldPar :: Int -> Bool+shouldPar = shouldParallelize parThreshold++{- | The additive moment sums of two columns, each an @nGroups@-length column:+@(n, Sx, Sy, Sxx, Syy, Sxy)@.+-}+data Moments = Moments+ { mN :: Column+ , mSx :: Column+ , mSy :: Column+ , mSxx :: Column+ , mSyy :: Column+ , mSxy :: Column+ }++{- | One pass over two Double-coercible columns @x@ and @y@ filling the count and+five sums, collapsing the Q9 regression family's six folds into a single pass.+'Nothing' unless both columns are non-null unboxed Int/Double.+-}+momentScatter :: VU.Vector Int -> Int -> Column -> Column -> Maybe Moments+momentScatter g nGroups colX colY = do+ xs <- cleanDoubleVector colX+ ys <- cleanDoubleVector colY+ let (cnt, sx, sy, sxx, syy, sxy) = momentPass g nGroups xs ys+ pure+ Moments+ { mN = fromUnboxedVector cnt+ , mSx = fromUnboxedVector sx+ , mSy = fromUnboxedVector sy+ , mSxx = fromUnboxedVector sxx+ , mSyy = fromUnboxedVector syy+ , mSxy = fromUnboxedVector sxy+ }++momentPass ::+ VU.Vector Int ->+ Int ->+ VU.Vector Double ->+ VU.Vector Double ->+ ( VU.Vector Int+ , VU.Vector Double+ , VU.Vector Double+ , VU.Vector Double+ , VU.Vector Double+ , VU.Vector Double+ )+momentPass g nGroups xs ys = runST $ do+ cnt <- VUM.replicate nGroups (0 :: Int)+ sx <- VUM.replicate nGroups (0 :: Double)+ sy <- VUM.replicate nGroups (0 :: Double)+ sxx <- VUM.replicate nGroups (0 :: Double)+ syy <- VUM.replicate nGroups (0 :: Double)+ sxy <- VUM.replicate nGroups (0 :: Double)+ let n = VU.length xs+ bump arr k d = VUM.unsafeRead arr k >>= \c -> VUM.unsafeWrite arr k (c + d)+ go !i+ | i >= n = pure ()+ | otherwise = do+ let !k = VU.unsafeIndex g i+ !x = VU.unsafeIndex xs i+ !y = VU.unsafeIndex ys i+ VUM.unsafeRead cnt k >>= \c -> VUM.unsafeWrite cnt k (c + 1)+ bump sx k x+ bump sy k y+ bump sxx k (x * x)+ bump syy k (y * y)+ bump sxy k (x * y)+ go (i + 1)+ go 0+ (,,,,,)+ <$> VU.unsafeFreeze cnt+ <*> VU.unsafeFreeze sx+ <*> VU.unsafeFreeze sy+ <*> VU.unsafeFreeze sxx+ <*> VU.unsafeFreeze syy+ <*> VU.unsafeFreeze sxy++{- | Parallel counterpart of 'momentScatter': one fused pass over both columns,+each group's six sums accumulated within one worker's range. Byte-identical to+'momentScatter'. 'Nothing' unless both columns are non-null unboxed Int/Double.+-}+momentScatterPar ::+ VU.Vector Int -> VU.Vector Int -> Int -> Column -> Column -> Maybe Moments+momentScatterPar vis offs nGroups colX colY+ | not (shouldPar (VU.length vis)) || nGroups <= 1 =+ momentScatter (rtgFromVis vis offs nGroups) nGroups colX colY+ | otherwise = do+ xs <- cleanDoubleVector colX+ ys <- cleanDoubleVector colY+ let !caps = capabilities+ !bounds = groupRangeBounds offs nGroups caps+ pure (unsafePerformIO (momentPar vis offs nGroups xs ys caps bounds))+{-# NOINLINE momentScatterPar #-}++-------------------------------------------------------------------------------++momentPar ::+ VU.Vector Int ->+ VU.Vector Int ->+ Int ->+ VU.Vector Double ->+ VU.Vector Double ->+ Int ->+ VU.Vector Int ->+ IO Moments+momentPar vis offs nGroups xs ys caps bounds = do+ cnt <- VUM.replicate nGroups (0 :: Int)+ sx <- VUM.replicate nGroups (0 :: Double)+ sy <- VUM.replicate nGroups (0 :: Double)+ sxx <- VUM.replicate nGroups (0 :: Double)+ syy <- VUM.replicate nGroups (0 :: Double)+ sxy <- VUM.replicate nGroups (0 :: Double)+ parallelBounds_ caps bounds $ \gs ge ->+ -- The six running sums carried in registers per group, written once.+ let grp !g+ | g >= ge = pure ()+ | otherwise = do+ let !e = VU.unsafeIndex offs (g + 1)+ inner !pos !ax !ay !axx !ayy !axy+ | pos >= e = do+ VUM.unsafeWrite sx g ax+ VUM.unsafeWrite sy g ay+ VUM.unsafeWrite sxx g axx+ VUM.unsafeWrite syy g ayy+ VUM.unsafeWrite sxy g axy+ | otherwise =+ let !row = VU.unsafeIndex vis pos+ !x = VU.unsafeIndex xs row+ !y = VU.unsafeIndex ys row+ in inner+ (pos + 1)+ (ax + x)+ (ay + y)+ (axx + x * x)+ (ayy + y * y)+ (axy + x * y)+ !s0 = VU.unsafeIndex offs g+ VUM.unsafeWrite cnt g (e - s0)+ inner s0 0 0 0 0 0+ grp (g + 1)+ in grp gs+ Moments . fromUnboxedVector+ <$> VU.unsafeFreeze cnt+ <*> (fromUnboxedVector <$> VU.unsafeFreeze sx)+ <*> (fromUnboxedVector <$> VU.unsafeFreeze sy)+ <*> (fromUnboxedVector <$> VU.unsafeFreeze sxx)+ <*> (fromUnboxedVector <$> VU.unsafeFreeze syy)+ <*> (fromUnboxedVector <$> VU.unsafeFreeze sxy)++-------------------------------------------------------------------------------+-- Streaming fused two-column moments (Q9)+-------------------------------------------------------------------------------++{- | Streaming counterpart of 'momentScatterPar': one fused pass over+@rowToGroup@ and the two TYPED value columns (Int values convert to Double+in-register — bit-identical to the @VU.map fromIntegral@ materialization it+replaces, with no 800MB intermediate column and no sequential conversion pass).+Each worker accumulates the six per-group sums over its contiguous row chunk in+original row order; partials merge in fixed worker order (counts exactly, the+five Double sums in chunk-major float order — deterministic at a fixed @-N@,+but a different summation order than the per-group gather kernel).+'Nothing' above 'streamGroupCap' or unless both columns are clean unboxed+Int/Double; the caller then keeps the gather path.+-}+momentStreamPar :: VU.Vector Int -> Int -> Column -> Column -> Maybe Moments+momentStreamPar rtg nGroups colX colY+ | nGroups <= 0 || nGroups > streamGroupCap = Nothing+ | otherwise = case (colX, colY) of+ ( UnboxedColumn Nothing (vx :: VU.Vector x)+ , UnboxedColumn Nothing (vy :: VU.Vector y)+ )+ | Just Refl <- testEquality (typeRep @x) (typeRep @Int)+ , Just Refl <- testEquality (typeRep @y) (typeRep @Int) ->+ Just (momentStreamII rtg nGroups vx vy)+ | Just Refl <- testEquality (typeRep @x) (typeRep @Int)+ , Just Refl <- testEquality (typeRep @y) (typeRep @Double) ->+ Just (momentStreamID rtg nGroups vx vy)+ | Just Refl <- testEquality (typeRep @x) (typeRep @Double)+ , Just Refl <- testEquality (typeRep @y) (typeRep @Int) ->+ Just (momentStreamDI rtg nGroups vx vy)+ | Just Refl <- testEquality (typeRep @x) (typeRep @Double)+ , Just Refl <- testEquality (typeRep @y) (typeRep @Double) ->+ Just (momentStreamDD rtg nGroups vx vy)+ _ -> Nothing+{-# NOINLINE momentStreamPar #-}++{- | Monomorphic entry points (see 'reduceParInt' for why the 'testEquality'+dispatch needs them).+-}+momentStreamII ::+ VU.Vector Int -> Int -> VU.Vector Int -> VU.Vector Int -> Moments+momentStreamII = momentStreamTyped+{-# NOINLINE momentStreamII #-}++momentStreamID ::+ VU.Vector Int -> Int -> VU.Vector Int -> VU.Vector Double -> Moments+momentStreamID = momentStreamTyped+{-# NOINLINE momentStreamID #-}++momentStreamDI ::+ VU.Vector Int -> Int -> VU.Vector Double -> VU.Vector Int -> Moments+momentStreamDI = momentStreamTyped+{-# NOINLINE momentStreamDI #-}++momentStreamDD ::+ VU.Vector Int -> Int -> VU.Vector Double -> VU.Vector Double -> Moments+momentStreamDD = momentStreamTyped+{-# NOINLINE momentStreamDD #-}++-- | The six per-group running sums of one worker chunk.+data MomentAcc = MomentAcc+ { maCnt :: !(VUM.IOVector Int)+ , maSx :: !(VUM.IOVector Double)+ , maSy :: !(VUM.IOVector Double)+ , maSxx :: !(VUM.IOVector Double)+ , maSyy :: !(VUM.IOVector Double)+ , maSxy :: !(VUM.IOVector Double)+ }++newMomentAcc :: Int -> IO MomentAcc+newMomentAcc nGroups =+ MomentAcc+ <$> VUM.replicate nGroups 0+ <*> VUM.replicate nGroups 0+ <*> VUM.replicate nGroups 0+ <*> VUM.replicate nGroups 0+ <*> VUM.replicate nGroups 0+ <*> VUM.replicate nGroups 0++momentStreamTyped ::+ forall a b.+ (VU.Unbox a, VU.Unbox b, Real a, Real b) =>+ VU.Vector Int ->+ Int ->+ VU.Vector a ->+ VU.Vector b ->+ Moments+{- The SPECIALIZE pragmas matter for the same reason as 'reduceParTyped': the+per-element @realToFrac@ must rewrite to @int2Double@/@id@ at a concrete type+or it goes through 'Rational' at runtime. -}+{-# SPECIALIZE momentStreamTyped ::+ VU.Vector Int -> Int -> VU.Vector Int -> VU.Vector Int -> Moments+ #-}+{-# SPECIALIZE momentStreamTyped ::+ VU.Vector Int -> Int -> VU.Vector Int -> VU.Vector Double -> Moments+ #-}+{-# SPECIALIZE momentStreamTyped ::+ VU.Vector Int -> Int -> VU.Vector Double -> VU.Vector Int -> Moments+ #-}+{-# SPECIALIZE momentStreamTyped ::+ VU.Vector Int -> Int -> VU.Vector Double -> VU.Vector Double -> Moments+ #-}+momentStreamTyped rtg nGroups vx vy = unsafePerformIO $ do+ let !n = VU.length rtg+ !caps' = if shouldPar n then capabilities else 1+ !per = (max 1 n + caps' - 1) `div` caps'+ parts <-+ forkJoin+ [ momentStreamChunk rtg nGroups vx vy lo hi+ | w <- [0 .. caps' - 1]+ , let lo = min n (w * per)+ , let hi = min n (lo + per)+ ]+ case parts of+ [] -> error "momentStreamTyped: no partials"+ (p0 : rest) -> do+ _ <-+ forkJoin+ [ mapM_ (\p -> mergeMomentRange p0 p lo hi) rest+ | (lo, hi) <- groupSlices nGroups+ ]+ freezeMoments p0+{-# INLINEABLE momentStreamTyped #-}++momentStreamChunk ::+ (VU.Unbox a, VU.Unbox b, Real a, Real b) =>+ VU.Vector Int ->+ Int ->+ VU.Vector a ->+ VU.Vector b ->+ Int ->+ Int ->+ IO MomentAcc+momentStreamChunk rtg nGroups vx vy lo hi = do+ acc@(MomentAcc cnt sx sy sxx syy sxy) <- newMomentAcc nGroups+ let go !i+ | i >= hi = pure ()+ | otherwise = do+ let !k = VU.unsafeIndex rtg i+ !x = realToFrac (VU.unsafeIndex vx i) :: Double+ !y = realToFrac (VU.unsafeIndex vy i) :: Double+ c <- VUM.unsafeRead cnt k+ VUM.unsafeWrite cnt k (c + 1)+ ax <- VUM.unsafeRead sx k+ VUM.unsafeWrite sx k (ax + x)+ ay <- VUM.unsafeRead sy k+ VUM.unsafeWrite sy k (ay + y)+ axx <- VUM.unsafeRead sxx k+ VUM.unsafeWrite sxx k (axx + x * x)+ ayy <- VUM.unsafeRead syy k+ VUM.unsafeWrite syy k (ayy + y * y)+ axy <- VUM.unsafeRead sxy k+ VUM.unsafeWrite sxy k (axy + x * y)+ go (i + 1)+ go lo+ pure acc+{-# INLINE momentStreamChunk #-}++mergeMomentRange :: MomentAcc -> MomentAcc -> Int -> Int -> IO ()+mergeMomentRange a b lo hi = go lo+ where+ go !g+ | g >= hi = pure ()+ | otherwise = do+ addI (maCnt a) (maCnt b) g+ addD (maSx a) (maSx b) g+ addD (maSy a) (maSy b) g+ addD (maSxx a) (maSxx b) g+ addD (maSyy a) (maSyy b) g+ addD (maSxy a) (maSxy b) g+ go (g + 1)+ addI p q g = do+ x <- VUM.unsafeRead p g+ y <- VUM.unsafeRead q g+ VUM.unsafeWrite p g (x + y)+ addD p q g = do+ x <- VUM.unsafeRead p g+ y <- VUM.unsafeRead q g+ VUM.unsafeWrite p g (x + y)++freezeMoments :: MomentAcc -> IO Moments+freezeMoments (MomentAcc cnt sx sy sxx syy sxy) =+ Moments . fromUnboxedVector+ <$> VU.unsafeFreeze cnt+ <*> (fromUnboxedVector <$> VU.unsafeFreeze sx)+ <*> (fromUnboxedVector <$> VU.unsafeFreeze sy)+ <*> (fromUnboxedVector <$> VU.unsafeFreeze sxx)+ <*> (fromUnboxedVector <$> VU.unsafeFreeze syy)+ <*> (fromUnboxedVector <$> VU.unsafeFreeze sxy)
+ src-internal/DataFrame/Internal/Aggregation/Kernel/Scatter.hs view
@@ -0,0 +1,818 @@+{-# LANGUAGE BangPatterns #-}+{-# LANGUAGE ExplicitNamespaces #-}+{-# LANGUAGE FlexibleContexts #-}+{-# LANGUAGE GADTs #-}+{-# LANGUAGE ScopedTypeVariables #-}+{-# LANGUAGE TypeApplications #-}++{- | The group-range scatter-accumulate reduction kernel: reduces a value column+over the grouped layout @(valueIndices, offsets)@.++Sequential and parallel are the same algorithm at different row counts, so they+live together. 'scatterReducePar' cuts the GROUP axis into @caps@ ranges of+roughly equal row count and lets workers write disjoint slots of one shared+output — no per-worker accumulator, no merge — which keeps each group's+accumulation order identical to 'scatterReduce' and the results byte-identical+at any @-N@. Below 'parThreshold' it delegates to 'scatterReduce' directly.++Contrast "DataFrame.Internal.Aggregation.Kernel.Dense", which scatters off+@rowToGroup@ with no gather but needs a small dense group domain.+-}+module DataFrame.Internal.Aggregation.Kernel.Scatter (+ scatterReduce,+ scatterReducePar,+ maxMinusMinScatterPar,+ top2SndScatter,++ -- * Group-range helpers+ -- $shared+ groupRangeBounds,+ rtgFromVis,+ overGroupsAcc,+ groupSlices,+ streamGroupCap,+) where++import Control.Monad (when)+import Control.Monad.ST (ST, runST)+import Data.Type.Equality (TestEquality (..), type (:~:) (Refl))+import qualified Data.Vector.Unboxed as VU+import qualified Data.Vector.Unboxed.Mutable as VUM+import System.IO.Unsafe (unsafePerformIO)+import Type.Reflection (typeRep)++import DataFrame.Internal.Aggregation.Reduction (Reduction (..))+import DataFrame.Internal.Column (+ Column (..),+ Columnable,+ fromUnboxedVector,+ materializePacked,+ )+import DataFrame.Internal.Control.Concurrent (+ capabilities,+ chunksFor,+ parThreshold,+ parallelBounds_,+ shouldParallelize,+ )++{- $shared+Also used by "DataFrame.Internal.Aggregation.Kernel.Moments", which partitions+the group axis the same way.+-}++-- | Whether to fan out at this row count.+shouldPar :: Int -> Bool+shouldPar = shouldParallelize parThreshold++scatterReduce ::+ Reduction -> VU.Vector Int -> Int -> Column -> Maybe Column+scatterReduce red g nGroups col = case col of+ UnboxedColumn Nothing (v :: VU.Vector a) ->+ case testEquality (typeRep @a) (typeRep @Int) of+ Just Refl -> Just (reduceTyped red g nGroups v intIdent)+ Nothing -> case testEquality (typeRep @a) (typeRep @Double) of+ Just Refl -> Just (reduceTyped red g nGroups v dblIdent)+ Nothing -> Nothing+ p@(PackedText _ _) -> scatterReduce red g nGroups (materializePacked p)+ _ -> Nothing+{-# INLINEABLE scatterReduce #-}++-- | Per-type seed identities for the order-preserving reductions.+data Idents a = Idents {minSeed :: !a, maxSeed :: !a}++intIdent :: Idents Int+intIdent = Idents maxBound minBound++dblIdent :: Idents Double+dblIdent = Idents (1 / 0) (negate (1 / 0))++reduceTyped ::+ forall a.+ (Columnable a, VU.Unbox a, Num a, Ord a, Real a) =>+ Reduction -> VU.Vector Int -> Int -> VU.Vector a -> Idents a -> Column+reduceTyped red g nGroups v idents = case red of+ RCount -> fromUnboxedVector (countScatter g nGroups)+ RSum -> fromUnboxedVector (sumScatter g nGroups v)+ RMin -> fromUnboxedVector (extremaScatter min (minSeed idents) g nGroups v)+ RMax -> fromUnboxedVector (extremaScatter max (maxSeed idents) g nGroups v)+ RMean -> fromUnboxedVector (meanScatter g nGroups v)+ RVar -> fromUnboxedVector (varScatter False g nGroups v)+ RStd -> fromUnboxedVector (varScatter True g nGroups v)+ RTop2Sum -> fromUnboxedVector (top2Scatter g nGroups v)+ RTop2Snd -> fromUnboxedVector (top2SndScatter g nGroups v)+{-# INLINE reduceTyped #-}++countScatter :: VU.Vector Int -> Int -> VU.Vector Int+countScatter g nGroups = runST $ do+ cnt <- VUM.replicate nGroups (0 :: Int)+ let n = VU.length g+ go !i+ | i >= n = pure ()+ | otherwise = do+ let !k = VU.unsafeIndex g i+ c <- VUM.unsafeRead cnt k+ VUM.unsafeWrite cnt k (c + 1)+ go (i + 1)+ go 0+ VU.unsafeFreeze cnt++sumScatter ::+ (VU.Unbox a, Num a) => VU.Vector Int -> Int -> VU.Vector a -> VU.Vector a+sumScatter g nGroups v = runST $ do+ s <- VUM.replicate nGroups 0+ let n = VU.length v+ go !i+ | i >= n = pure ()+ | otherwise = do+ let !k = VU.unsafeIndex g i+ cur <- VUM.unsafeRead s k+ VUM.unsafeWrite s k (cur + VU.unsafeIndex v i)+ go (i + 1)+ go 0+ VU.unsafeFreeze s+{-# INLINE sumScatter #-}++extremaScatter ::+ (VU.Unbox a) =>+ (a -> a -> a) -> a -> VU.Vector Int -> Int -> VU.Vector a -> VU.Vector a+extremaScatter combine seed g nGroups v = runST $ do+ m <- VUM.replicate nGroups seed+ let n = VU.length v+ go !i+ | i >= n = pure ()+ | otherwise = do+ let !k = VU.unsafeIndex g i+ cur <- VUM.unsafeRead m k+ VUM.unsafeWrite m k (combine cur (VU.unsafeIndex v i))+ go (i + 1)+ go 0+ VU.unsafeFreeze m+{-# INLINE extremaScatter #-}++meanScatter ::+ (VU.Unbox a, Real a) => VU.Vector Int -> Int -> VU.Vector a -> VU.Vector Double+meanScatter g nGroups v = runST $ do+ s <- VUM.replicate nGroups (0 :: Double)+ cnt <- VUM.replicate nGroups (0 :: Int)+ scatterSumCount g v s cnt+ finalizeMean nGroups s cnt+{-# INLINE meanScatter #-}++scatterSumCount ::+ (VU.Unbox a, Real a) =>+ VU.Vector Int ->+ VU.Vector a ->+ VUM.MVector s Double ->+ VUM.MVector s Int ->+ ST s ()+scatterSumCount g v s cnt = go 0+ where+ n = VU.length v+ go !i+ | i >= n = pure ()+ | otherwise = do+ let !k = VU.unsafeIndex g i+ !x = realToFrac (VU.unsafeIndex v i)+ curS <- VUM.unsafeRead s k+ VUM.unsafeWrite s k (curS + x)+ curC <- VUM.unsafeRead cnt k+ VUM.unsafeWrite cnt k (curC + 1)+ go (i + 1)+{-# INLINE scatterSumCount #-}++finalizeMean ::+ Int -> VUM.MVector s Double -> VUM.MVector s Int -> ST s (VU.Vector Double)+finalizeMean nGroups s cnt = do+ out <- VUM.new nGroups+ let go !k+ | k >= nGroups = pure ()+ | otherwise = do+ sv <- VUM.unsafeRead s k+ c <- VUM.unsafeRead cnt k+ VUM.unsafeWrite out k (if c == 0 then 0 / 0 else sv / fromIntegral c)+ go (k + 1)+ go 0+ VU.unsafeFreeze out++varScatter ::+ (VU.Unbox a, Real a) =>+ Bool -> VU.Vector Int -> Int -> VU.Vector a -> VU.Vector Double+varScatter takeSqrt g nGroups v = runST $ do+ cnt <- VUM.replicate nGroups (0 :: Int)+ meanV <- VUM.replicate nGroups (0 :: Double)+ m2 <- VUM.replicate nGroups (0 :: Double)+ let n = VU.length v+ go !i+ | i >= n = pure ()+ | otherwise = do+ let !k = VU.unsafeIndex g i+ !x = realToFrac (VU.unsafeIndex v i)+ c <- VUM.unsafeRead cnt k+ mu <- VUM.unsafeRead meanV k+ mm <- VUM.unsafeRead m2 k+ let !c' = c + 1+ !delta = x - mu+ !mu' = mu + delta / fromIntegral c'+ !mm' = mm + delta * (x - mu')+ VUM.unsafeWrite cnt k c'+ VUM.unsafeWrite meanV k mu'+ VUM.unsafeWrite m2 k mm'+ go (i + 1)+ go 0+ out <- VUM.new nGroups+ let fin !k+ | k >= nGroups = pure ()+ | otherwise = do+ c <- VUM.unsafeRead cnt k+ mm <- VUM.unsafeRead m2 k+ let var = if c < 2 then 0 else mm / fromIntegral (c - 1)+ VUM.unsafeWrite out k (if takeSqrt then sqrt var else var)+ fin (k + 1)+ fin 0+ VU.unsafeFreeze out+{-# INLINE varScatter #-}++top2Scatter ::+ (VU.Unbox a, Real a) => VU.Vector Int -> Int -> VU.Vector a -> VU.Vector Double+top2Scatter g nGroups v = runST $ do+ let ninf = negate (1 / 0) :: Double+ m1 <- VUM.replicate nGroups ninf+ m2 <- VUM.replicate nGroups ninf+ let n = VU.length v+ go !i+ | i >= n = pure ()+ | otherwise = do+ let !k = VU.unsafeIndex g i+ !x = realToFrac (VU.unsafeIndex v i)+ a1 <- VUM.unsafeRead m1 k+ if x > a1+ then do+ VUM.unsafeWrite m1 k x+ VUM.unsafeWrite m2 k a1+ else do+ a2 <- VUM.unsafeRead m2 k+ when (x > a2) (VUM.unsafeWrite m2 k x)+ go (i + 1)+ go 0+ out <- VUM.new nGroups+ let fin !k+ | k >= nGroups = pure ()+ | otherwise = do+ a1 <- VUM.unsafeRead m1 k+ a2 <- VUM.unsafeRead m2 k+ let s = (if isInfinite a1 then 0 else a1) + (if isInfinite a2 then 0 else a2)+ VUM.unsafeWrite out k s+ fin (k + 1)+ fin 0+ VU.unsafeFreeze out+{-# INLINE top2Scatter #-}++groupRangeBounds :: VU.Vector Int -> Int -> Int -> VU.Vector Int+groupRangeBounds offs nGroups caps = VU.create $ do+ b <- VUM.new (caps + 1)+ let !nRows = VU.unsafeIndex offs nGroups+ !per = max 1 ((nRows + caps - 1) `div` caps)+ adv !target !gg+ | gg >= nGroups = nGroups+ | VU.unsafeIndex offs gg >= target = gg+ | otherwise = adv target (gg + 1)+ go !w !prev+ | w >= caps = VUM.unsafeWrite b caps nGroups+ | otherwise = do+ let !target = min nRows (w * per)+ !g = adv target prev+ VUM.unsafeWrite b w g+ go (w + 1) g+ VUM.unsafeWrite b 0 0+ go 1 0+ pure b++scatterReducePar ::+ Reduction -> VU.Vector Int -> VU.Vector Int -> Int -> Column -> Maybe Column+scatterReducePar red vis offs nGroups col+ | not (shouldParallelize parThreshold (VU.length vis)) || nGroups <= 1 =+ scatterReduce red (rtgFromVis vis offs nGroups) nGroups col+ | otherwise = case col of+ UnboxedColumn Nothing (v :: VU.Vector a) ->+ case testEquality (typeRep @a) (typeRep @Int) of+ Just Refl -> Just (reduceParInt red vis offs nGroups v)+ Nothing -> case testEquality (typeRep @a) (typeRep @Double) of+ Just Refl -> Just (reduceParDouble red vis offs nGroups v)+ Nothing -> Nothing+ p@(PackedText _ _) -> scatterReducePar red vis offs nGroups (materializePacked p)+ _ -> Nothing+{-# NOINLINE scatterReducePar #-}++{- | Monomorphic entry points: the 'testEquality' dispatch above only yields an+unsafe coercion, so a direct call to the polymorphic 'reduceParTyped' there+would stay at the abstract element type and never meet its SPECIALIZE rules;+calling through these fixed-type wrappers (the coercion lands on the argument)+does.+-}+reduceParInt ::+ Reduction -> VU.Vector Int -> VU.Vector Int -> Int -> VU.Vector Int -> Column+reduceParInt red vis offs nGroups v = reduceParTyped red vis offs nGroups v intIdent++reduceParDouble ::+ Reduction -> VU.Vector Int -> VU.Vector Int -> Int -> VU.Vector Double -> Column+reduceParDouble red vis offs nGroups v = reduceParTyped red vis offs nGroups v dblIdent++rtgFromVis :: VU.Vector Int -> VU.Vector Int -> Int -> VU.Vector Int+rtgFromVis vis offs nGroups = VU.create $ do+ let n = VU.length vis+ rtg <- VUM.new (max 1 n)+ let go !g+ | g >= nGroups = pure ()+ | otherwise = do+ let !e = VU.unsafeIndex offs (g + 1)+ inner !pos+ | pos >= e = pure ()+ | otherwise = do+ VUM.unsafeWrite rtg (VU.unsafeIndex vis pos) g+ inner (pos + 1)+ inner (VU.unsafeIndex offs g)+ go (g + 1)+ go 0+ pure rtg++reduceParTyped ::+ forall a.+ (Columnable a, VU.Unbox a, Num a, Ord a, Real a) =>+ Reduction ->+ VU.Vector Int ->+ VU.Vector Int ->+ Int ->+ VU.Vector a ->+ Idents a ->+ Column+{- The SPECIALIZE pragmas matter: without them the @realToFrac@ in the+mean/var/top2 kernels survives to runtime as a dictionary call through+'Rational' (the Double->Double/Int->Double rewrite rules only fire once the+type is concrete), costing ~4x on the whole pass. -}+{-# SPECIALIZE reduceParTyped ::+ Reduction ->+ VU.Vector Int ->+ VU.Vector Int ->+ Int ->+ VU.Vector Int ->+ Idents Int ->+ Column+ #-}+{-# SPECIALIZE reduceParTyped ::+ Reduction ->+ VU.Vector Int ->+ VU.Vector Int ->+ Int ->+ VU.Vector Double ->+ Idents Double ->+ Column+ #-}+reduceParTyped red vis offs nGroups v idents =+ let !caps = capabilities+ !bounds = groupRangeBounds offs nGroups caps+ in case red of+ RCount -> fromUnboxedVector (unsafePerformIO (countPar vis offs nGroups caps bounds))+ RSum -> fromUnboxedVector (unsafePerformIO (sumPar vis offs nGroups v caps bounds))+ RMin ->+ fromUnboxedVector+ (unsafePerformIO (extremaPar min (minSeed idents) vis offs nGroups v caps bounds))+ RMax ->+ fromUnboxedVector+ (unsafePerformIO (extremaPar max (maxSeed idents) vis offs nGroups v caps bounds))+ RMean -> fromUnboxedVector (unsafePerformIO (meanPar vis offs nGroups v caps bounds))+ RVar ->+ fromUnboxedVector+ (unsafePerformIO (varPar False vis offs nGroups v caps bounds))+ RStd ->+ fromUnboxedVector (unsafePerformIO (varPar True vis offs nGroups v caps bounds))+ RTop2Sum -> fromUnboxedVector (unsafePerformIO (top2Par vis offs nGroups v caps bounds))+ RTop2Snd ->+ fromUnboxedVector (unsafePerformIO (top2SndPar vis offs nGroups v caps bounds))+{-# INLINEABLE reduceParTyped #-}++{- | For each group in @[gs, ge)@, fold the group's rows (in @valueIndices@+order, i.e. ascending original-row order) into an accumulator held in+registers, then hand the final accumulator to @done@ exactly once. Keeping the+running state out of memory leaves one write per group instead of a+read-modify-write per row; the per-group fold order is unchanged, so results+stay byte-identical to the row-wise variant.+-}+overGroupsAcc ::+ VU.Vector Int ->+ VU.Vector Int ->+ Int ->+ Int ->+ acc ->+ (acc -> Int -> acc) ->+ (Int -> acc -> IO ()) ->+ IO ()+overGroupsAcc vis offs gs ge seed step done = grp gs+ where+ grp !g+ | g >= ge = pure ()+ | otherwise = do+ let !e = VU.unsafeIndex offs (g + 1)+ inner !pos !acc+ | pos >= e = pure acc+ | otherwise = inner (pos + 1) (step acc (VU.unsafeIndex vis pos))+ acc <- inner (VU.unsafeIndex offs g) seed+ done g acc+ grp (g + 1)+{-# INLINE overGroupsAcc #-}++countPar ::+ VU.Vector Int ->+ VU.Vector Int ->+ Int ->+ Int ->+ VU.Vector Int ->+ IO (VU.Vector Int)+countPar _vis offs nGroups caps bounds = do+ out <- VUM.replicate nGroups (0 :: Int)+ parallelBounds_ caps bounds $ \gs ge ->+ let grp !g+ | g >= ge = pure ()+ | otherwise = do+ let !c = VU.unsafeIndex offs (g + 1) - VU.unsafeIndex offs g+ VUM.unsafeWrite out g c+ grp (g + 1)+ in grp gs+ VU.unsafeFreeze out++sumPar ::+ (VU.Unbox a, Num a) =>+ VU.Vector Int ->+ VU.Vector Int ->+ Int ->+ VU.Vector a ->+ Int ->+ VU.Vector Int ->+ IO (VU.Vector a)+sumPar vis offs nGroups v caps bounds = do+ out <- VUM.replicate nGroups 0+ parallelBounds_ caps bounds $ \gs ge ->+ overGroupsAcc vis offs gs ge 0 (\acc row -> acc + VU.unsafeIndex v row) $+ VUM.unsafeWrite out+ VU.unsafeFreeze out+{-# INLINE sumPar #-}++extremaPar ::+ (VU.Unbox a) =>+ (a -> a -> a) ->+ a ->+ VU.Vector Int ->+ VU.Vector Int ->+ Int ->+ VU.Vector a ->+ Int ->+ VU.Vector Int ->+ IO (VU.Vector a)+extremaPar combine seed vis offs nGroups v caps bounds = do+ out <- VUM.replicate nGroups seed+ parallelBounds_ caps bounds $ \gs ge ->+ overGroupsAcc+ vis+ offs+ gs+ ge+ seed+ (\acc row -> combine acc (VU.unsafeIndex v row))+ $ VUM.unsafeWrite out+ VU.unsafeFreeze out+{-# INLINE extremaPar #-}++meanPar ::+ (VU.Unbox a, Real a) =>+ VU.Vector Int ->+ VU.Vector Int ->+ Int ->+ VU.Vector a ->+ Int ->+ VU.Vector Int ->+ IO (VU.Vector Double)+meanPar vis offs nGroups v caps bounds = do+ out <- VUM.replicate nGroups (0 :: Double)+ parallelBounds_ caps bounds $ \gs ge ->+ let grp !g+ | g >= ge = pure ()+ | otherwise = do+ let !e = VU.unsafeIndex offs (g + 1)+ inner !pos !acc+ | pos >= e = acc+ | otherwise =+ inner+ (pos + 1)+ (acc + realToFrac (VU.unsafeIndex v (VU.unsafeIndex vis pos)))+ !s0 = VU.unsafeIndex offs g+ !total = inner s0 0+ !c = e - s0+ VUM.unsafeWrite out g (if c == 0 then 0 / 0 else total / fromIntegral c)+ grp (g + 1)+ in grp gs+ VU.unsafeFreeze out+{-# INLINE meanPar #-}++varPar ::+ (VU.Unbox a, Real a) =>+ Bool ->+ VU.Vector Int ->+ VU.Vector Int ->+ Int ->+ VU.Vector a ->+ Int ->+ VU.Vector Int ->+ IO (VU.Vector Double)+varPar takeSqrt vis offs nGroups v caps bounds = do+ out <- VUM.replicate nGroups (0 :: Double)+ parallelBounds_ caps bounds $ \gs ge ->+ -- Per-group Welford state (count, mean, M2) carried in registers; the+ -- update order per group is the same ascending row order as before.+ let grp !g+ | g >= ge = pure ()+ | otherwise = do+ let !e = VU.unsafeIndex offs (g + 1)+ inner !pos !c !mu !mm+ | pos >= e =+ let var = if c < 2 then 0 else mm / fromIntegral (c - 1)+ in if takeSqrt then sqrt var else var+ | otherwise =+ let !x = realToFrac (VU.unsafeIndex v (VU.unsafeIndex vis pos))+ !c' = c + 1+ !delta = x - mu+ !mu' = mu + delta / fromIntegral c'+ !mm' = mm + delta * (x - mu')+ in inner (pos + 1) c' mu' mm'+ !res = inner (VU.unsafeIndex offs g) (0 :: Int) 0 0+ VUM.unsafeWrite out g res+ grp (g + 1)+ in grp gs+ VU.unsafeFreeze out+{-# INLINE varPar #-}++top2Par ::+ (VU.Unbox a, Real a) =>+ VU.Vector Int ->+ VU.Vector Int ->+ Int ->+ VU.Vector a ->+ Int ->+ VU.Vector Int ->+ IO (VU.Vector Double)+top2Par vis offs nGroups v caps bounds = do+ let ninf = negate (1 / 0) :: Double+ out <- VUM.replicate nGroups (0 :: Double)+ parallelBounds_ caps bounds $ \gs ge ->+ -- The (largest, second-largest) pair carried in registers per group.+ let grp !g+ | g >= ge = pure ()+ | otherwise = do+ let !e = VU.unsafeIndex offs (g + 1)+ inner !pos !a1 !a2+ | pos >= e =+ (if isInfinite a1 then 0 else a1)+ + (if isInfinite a2 then 0 else a2)+ | otherwise =+ let !x = realToFrac (VU.unsafeIndex v (VU.unsafeIndex vis pos))+ in if x > a1+ then inner (pos + 1) x a1+ else inner (pos + 1) a1 (max a2 x)+ !res = inner (VU.unsafeIndex offs g) ninf ninf+ VUM.unsafeWrite out g res+ grp (g + 1)+ in grp gs+ VU.unsafeFreeze out+{-# INLINE top2Par #-}++{- | Second-largest value per group: the same (largest, second-largest)+register pair as 'top2Par', finalized to the second max alone. Size-1 groups+finalize the @-inf@ seed to NaN (documented; see+'DataFrame.Internal.AggKernel.top2SndScatter').+-}+top2SndPar ::+ (VU.Unbox a, Real a) =>+ VU.Vector Int ->+ VU.Vector Int ->+ Int ->+ VU.Vector a ->+ Int ->+ VU.Vector Int ->+ IO (VU.Vector Double)+top2SndPar vis offs nGroups v caps bounds = do+ let ninf = negate (1 / 0) :: Double+ out <- VUM.replicate nGroups (0 :: Double)+ parallelBounds_ caps bounds $ \gs ge ->+ -- The (largest, second-largest) pair carried in registers per group.+ let grp !g+ | g >= ge = pure ()+ | otherwise = do+ let !e = VU.unsafeIndex offs (g + 1)+ inner !pos !a1 !a2+ | pos >= e = if isInfinite a2 then 0 / 0 else a2+ | otherwise =+ let !x = realToFrac (VU.unsafeIndex v (VU.unsafeIndex vis pos))+ in if x > a1+ then inner (pos + 1) x a1+ else inner (pos + 1) a1 (max a2 x)+ !res = inner (VU.unsafeIndex offs g) ninf ninf+ VUM.unsafeWrite out g res+ grp (g + 1)+ in grp gs+ VU.unsafeFreeze out+{-# INLINE top2SndPar #-}++-------------------------------------------------------------------------------+-- Parallel fused max(a) - min(b) (Q7 at wide group domains)+-------------------------------------------------------------------------------++{- | Fused @max a - min b@ over the group-range layout: ONE traversal of+@valueIndices@ accumulating both extrema, parallel by disjoint group range with+no cross-worker merge. min/max are order-independent, so the result is+byte-identical to running the two gather extrema passes separately; the fusion+halves the index traffic. 'Nothing' below the parallel threshold or unless both+columns are clean unboxed and same-typed (Int/Int keeps the Int result of the+interpreter; Double/Double the Double one) — the caller then keeps its two-pass+fallback.+-}+maxMinusMinScatterPar ::+ VU.Vector Int -> VU.Vector Int -> Int -> Column -> Column -> Maybe Column+maxMinusMinScatterPar vis offs nGroups ca cb+ | not (shouldPar (VU.length vis)) || nGroups <= 1 = Nothing+ | otherwise = case (ca, cb) of+ ( UnboxedColumn Nothing (va :: VU.Vector x)+ , UnboxedColumn Nothing (vb :: VU.Vector y)+ )+ | Just Refl <- testEquality (typeRep @x) (typeRep @Int)+ , Just Refl <- testEquality (typeRep @y) (typeRep @Int) ->+ Just (maxMinusMinParInt vis offs nGroups va vb caps bounds)+ | Just Refl <- testEquality (typeRep @x) (typeRep @Double)+ , Just Refl <- testEquality (typeRep @y) (typeRep @Double) ->+ Just (maxMinusMinParDbl vis offs nGroups va vb caps bounds)+ _ -> Nothing+ where+ !caps = capabilities+ !bounds = groupRangeBounds offs nGroups caps+{-# NOINLINE maxMinusMinScatterPar #-}++{- | Monomorphic entry points (see 'reduceParInt' for why the 'testEquality'+dispatch needs them).+-}+maxMinusMinParInt ::+ VU.Vector Int ->+ VU.Vector Int ->+ Int ->+ VU.Vector Int ->+ VU.Vector Int ->+ Int ->+ VU.Vector Int ->+ Column+maxMinusMinParInt vis offs nGroups va vb caps bounds =+ fromUnboxedVector+ ( unsafePerformIO+ (maxMinusMinPar minBound maxBound vis offs nGroups va vb caps bounds)+ )+{-# NOINLINE maxMinusMinParInt #-}++maxMinusMinParDbl ::+ VU.Vector Int ->+ VU.Vector Int ->+ Int ->+ VU.Vector Double ->+ VU.Vector Double ->+ Int ->+ VU.Vector Int ->+ Column+maxMinusMinParDbl vis offs nGroups va vb caps bounds =+ fromUnboxedVector+ ( unsafePerformIO+ (maxMinusMinPar (negate (1 / 0)) (1 / 0) vis offs nGroups va vb caps bounds)+ )+{-# NOINLINE maxMinusMinParDbl #-}++maxMinusMinPar ::+ (VU.Unbox a, Num a, Ord a) =>+ a ->+ a ->+ VU.Vector Int ->+ VU.Vector Int ->+ Int ->+ VU.Vector a ->+ VU.Vector a ->+ Int ->+ VU.Vector Int ->+ IO (VU.Vector a)+{-# SPECIALIZE maxMinusMinPar ::+ Int ->+ Int ->+ VU.Vector Int ->+ VU.Vector Int ->+ Int ->+ VU.Vector Int ->+ VU.Vector Int ->+ Int ->+ VU.Vector Int ->+ IO (VU.Vector Int)+ #-}+{-# SPECIALIZE maxMinusMinPar ::+ Double ->+ Double ->+ VU.Vector Int ->+ VU.Vector Int ->+ Int ->+ VU.Vector Double ->+ VU.Vector Double ->+ Int ->+ VU.Vector Int ->+ IO (VU.Vector Double)+ #-}+maxMinusMinPar maxSeed minSeed vis offs nGroups va vb caps bounds = do+ out <- VUM.new nGroups+ parallelBounds_ caps bounds $ \gs ge ->+ -- Both extrema carried in registers per group; one traversal of the+ -- shared index slice reads both value columns.+ let grp !g+ | g >= ge = pure ()+ | otherwise = do+ let !e = VU.unsafeIndex offs (g + 1)+ inner !pos !mx !mn+ | pos >= e = mx - mn+ | otherwise =+ let !row = VU.unsafeIndex vis pos+ in inner+ (pos + 1)+ (max mx (VU.unsafeIndex va row))+ (min mn (VU.unsafeIndex vb row))+ !res = inner (VU.unsafeIndex offs g) maxSeed minSeed+ VUM.unsafeWrite out g res+ grp (g + 1)+ in grp gs+ VU.unsafeFreeze out++-------------------------------------------------------------------------------+-- Streaming (rowToGroup-scatter) kernels: no valueIndices, no placement pass+-------------------------------------------------------------------------------++{- | Group-count cap for the FUSED streaming rtg-scatter kernels+('DataFrame.Internal.Aggregation.Kernel.Moments.momentStreamPar',+'DataFrame.Internal.Aggregation.Kernel.Fused.runFusedAggs'). Above+'DataFrame.Internal.Grouping.Direct.directThreshold' the per-worker accumulator+arrays overflow cache, so a SINGLE streaming reduction loses to a gather pass —+the per-expression dispatch keeps that threshold. A fused multi-reduction pass+amortizes those misses across all its reductions AND avoids the deferred+@valueIndices@ placement entirely, which flips the comparison (measured at 1e6+groups / 1e8 rows on -N16: rowToGroup 0.4s + fused 3-sum stream 1.4s, against+placement 1.1s + fused gather 1.05s), so the fused cap extends to+'directGroupThreshold' — every direct-grouped frame can stream. Wider+groupings are necessarily hash-path (eager @valueIndices@) and use the fused+GATHER kernel ('DataFrame.Internal.Aggregation.Kernel.Fused.runGatherAggs')+instead. Memory: @capabilities * nGroups@ words per accumulator array, at most+~128MB transient at -N16.+-}+streamGroupCap :: Int+streamGroupCap = 1048576++{- | Near-equal contiguous slices of the group domain for parallel merges.+Below 4096 groups (or single-capability) the merge stays on one thread.+-}+groupSlices :: Int -> [(Int, Int)]+groupSlices = chunksFor 4096++{- | Second-largest value per group: the same (largest, second-largest)+accumulator pair as 'top2Scatter', but the finalize returns the second max+alone. A group of size 1 (or 0) leaves the @-inf@ seed in the second slot, so+its output is NaN — documented behaviour (the db-benchmark Q8 data has no+size-1 @id6@ groups). Like 'top2Scatter''s @-inf -> 0@ guard, an actual+infinite data value in the second slot is indistinguishable from the seed.+-}+top2SndScatter ::+ (VU.Unbox a, Real a) => VU.Vector Int -> Int -> VU.Vector a -> VU.Vector Double+top2SndScatter g nGroups v = runST $ do+ let ninf = negate (1 / 0) :: Double+ m1 <- VUM.replicate nGroups ninf+ m2 <- VUM.replicate nGroups ninf+ let n = VU.length v+ go !i+ | i >= n = pure ()+ | otherwise = do+ let !k = VU.unsafeIndex g i+ !x = realToFrac (VU.unsafeIndex v i)+ a1 <- VUM.unsafeRead m1 k+ if x > a1+ then do+ VUM.unsafeWrite m1 k x+ VUM.unsafeWrite m2 k a1+ else do+ a2 <- VUM.unsafeRead m2 k+ when (x > a2) (VUM.unsafeWrite m2 k x)+ go (i + 1)+ go 0+ out <- VUM.new nGroups+ let fin !k+ | k >= nGroups = pure ()+ | otherwise = do+ a2 <- VUM.unsafeRead m2 k+ VUM.unsafeWrite out k (if isInfinite a2 then 0 / 0 else a2)+ fin (k + 1)+ fin 0+ VU.unsafeFreeze out+{-# INLINE top2SndScatter #-}
+ src-internal/DataFrame/Internal/Aggregation/Plan.hs view
@@ -0,0 +1,222 @@+{-# LANGUAGE AllowAmbiguousTypes #-}+{-# LANGUAGE ExplicitNamespaces #-}+{-# LANGUAGE GADTs #-}+{-# LANGUAGE LambdaCase #-}+{-# LANGUAGE OverloadedStrings #-}+{-# LANGUAGE ScopedTypeVariables #-}+{-# LANGUAGE TypeApplications #-}++{- | The aggregation fast-path planner. 'planAgg' recognises a supported+aggregate shape over a clean unboxed Int/Double column and returns an 'AggPlan';+'planMoments' recognises the six-fold regression shape and returns a+'MomentPlan'. Planning only — the kernels live under "Kernel".+-}+module DataFrame.Internal.Aggregation.Plan (+ AggPlan (..),+ planAgg,+ MomentPlan (..),+ planMoments,+) where++import qualified Data.Map.Strict as M+import qualified Data.Text as T+import Data.Type.Equality (TestEquality (..), type (:~:) (Refl))+import qualified Data.Vector.Unboxed as VU++import DataFrame.Internal.Aggregation.Reduction (Reduction (..))+import DataFrame.Internal.Column (Column (..))+import DataFrame.Internal.DataFrame (+ DataFrame (derivingExpressions),+ GroupedDataFrame (..),+ getColumn,+ )+import DataFrame.Internal.Expression (+ AggStrategy (..),+ BinaryOp (binaryCommutative, binaryName),+ Expr (..),+ UExpr (..),+ )+import Type.Reflection (Typeable, typeRep)++{- | The plan 'planAgg' produces for a recognised output expression. The median+plan carries only the column name (the holistic grouped sort lives in the+operations layer, where @vector-algorithms@ is available).+-}+data AggPlan+ = -- | A single scatter reduction over one named column.+ PlanScatter Reduction T.Text+ | -- | @max a - min b@ (Q7): two scatters then a vectorized combine.+ PlanMaxMinusMin T.Text T.Text+ | -- | Holistic median over one named column.+ PlanMedian T.Text++{- | Inspect a named output expression; return @Just plan@ on a recognised shape+over a present clean column, else 'Nothing'. Nullable or non-Int/Double columns+are rejected here so the scatter only sees a clean unboxed vector.+-}+planAgg :: GroupedDataFrame -> UExpr -> Maybe AggPlan+planAgg gdf (UExpr (expr :: Expr a)) = case expr of+ Agg (FoldAgg tag _ _) (Col name) -> foldPlan tag name+ Agg (MergeAgg tag _ _ _ _) (Col name) -> mergePlan tag name+ Agg (CollectAgg tag _) (Col name) -> collectPlan tag name+ Binary+ op+ (Agg (FoldAgg lt Nothing _) (Col a))+ (Agg (FoldAgg rt Nothing _) (Col b)) ->+ if binaryName op == "sub" && lt == "maximum" && rt == "minimum"+ then requireBoth a b (PlanMaxMinusMin a b)+ else Nothing+ _ -> Nothing+ where+ foldPlan tag name = case tag of+ "sum" -> require name (PlanScatter RSum name)+ "minimum" -> require name (PlanScatter RMin name)+ "maximum" -> require name (PlanScatter RMax name)+ _ -> Nothing+ mergePlan tag name = case tag of+ "mean" -> outputType @Double >> require name (PlanScatter RMean name)+ "count" -> outputType @Int >> require name (PlanScatter RCount name)+ _ -> Nothing+ outputType :: forall t. (Typeable t) => Maybe ()+ outputType = case testEquality (typeRep @a) (typeRep @t) of+ Just Refl -> Just ()+ Nothing -> Nothing+ collectPlan tag name = case tag of+ "stddev" -> require name (PlanScatter RStd name)+ "variance" -> require name (PlanScatter RVar name)+ "top2Sum" -> require name (PlanScatter RTop2Sum name)+ "top2Snd" -> require name (PlanScatter RTop2Snd name)+ "median" -> require name (PlanMedian name)+ _ -> Nothing+ require name plan = colUnboxedNumeric name >> Just plan+ requireBoth a b plan = colUnboxedNumeric a >> colUnboxedNumeric b >> Just plan+ colUnboxedNumeric name = case getColumn name (fullDataframe gdf) of+ Just c | isUnboxedNumeric c -> Just ()+ _ -> Nothing++-- | The matcher only fires on non-null unboxed Int/Double columns.+isUnboxedNumeric :: Column -> Bool+isUnboxedNumeric = \case+ UnboxedColumn Nothing (_ :: VU.Vector a) ->+ case testEquality (typeRep @a) (typeRep @Int) of+ Just Refl -> True+ Nothing -> case testEquality (typeRep @a) (typeRep @Double) of+ Just Refl -> True+ Nothing -> False+ _ -> False++{- | A recognised moment (Q9 regression) aggregate group: six output columns that+form the sufficient statistics of two base columns @x@ and @y@. The caller runs+'momentScatter' once and binds each output name to a field of the result.+-}+data MomentPlan = MomentPlan+ { mpColX :: T.Text+ , mpColY :: T.Text+ , mpNName :: T.Text+ , mpSxName :: T.Text+ , mpSyName :: T.Text+ , mpSxxName :: T.Text+ , mpSyyName :: T.Text+ , mpSxyName :: T.Text+ }++{- | The shape of a sum's argument once unary coercions are peeled and derived+columns are resolved through @derivingExpressions@: either linear in one base+column or the product of two base columns (sorted).+-}+data Term+ = Lin T.Text+ | Prod T.Text T.Text+ deriving (Eq, Ord, Show)++{- | Recognise the moment shape across a whole @aggregate@ list: exactly+@count@, @sum(x)@, @sum(y)@, @sum(x*x)@, @sum(y*y)@, @sum(x*y)@ over two distinct+clean unboxed base columns. 'Nothing' on any other set.+-}+planMoments :: GroupedDataFrame -> [(T.Text, UExpr)] -> Maybe MomentPlan+planMoments gdf aggs+ | length aggs /= 6 = Nothing+ | otherwise = do+ let exprs = derivingExpressions (fullDataframe gdf)+ roles <- traverse (classify exprs) aggs+ let names = M.fromList [(r, nm) | (nm, r) <- roles]+ nName <- M.lookup RoleN names+ (x, y) <- pickBaseColumns roles+ sxName <- M.lookup (RoleLin x) names+ syName <- M.lookup (RoleLin y) names+ sxxName <- M.lookup (RoleProd x x) names+ syyName <- M.lookup (RoleProd y y) names+ sxyName <- M.lookup (RoleProd x y) names+ _ <- if x /= y then Just () else Nothing+ _ <- colUnboxedNumeric x+ _ <- colUnboxedNumeric y+ pure+ MomentPlan+ { mpColX = x+ , mpColY = y+ , mpNName = nName+ , mpSxName = sxName+ , mpSyName = syName+ , mpSxxName = sxxName+ , mpSyyName = syyName+ , mpSxyName = sxyName+ }+ where+ colUnboxedNumeric name = case getColumn name (fullDataframe gdf) of+ Just c | isUnboxedNumeric c -> Just ()+ _ -> Nothing++-- | The output role each named aggregation plays in the moment shape.+data Role+ = RoleN+ | RoleLin T.Text+ | RoleProd T.Text T.Text+ deriving (Eq, Ord, Show)++-- | Tag a single named aggregation with its moment role, or reject the group.+classify :: M.Map T.Text UExpr -> (T.Text, UExpr) -> Maybe (T.Text, Role)+classify exprs (name, UExpr expr) = case expr of+ Agg (MergeAgg "count" _ _ _ _) _ -> Just (name, RoleN)+ Agg (FoldAgg "sum" _ _) arg -> (\t -> (name, termRole t)) <$> resolveTerm exprs (UExpr arg)+ _ -> Nothing++termRole :: Term -> Role+termRole (Lin a) = RoleLin a+termRole (Prod a b) = RoleProd a b++{- | Resolve a (sum-argument) expression to its 'Term'. Peels @toDouble@-style+unary coercions, follows a derived column to its stored expression, and+recognises a commutative product of two linear terms.+-}+resolveTerm :: M.Map T.Text UExpr -> UExpr -> Maybe Term+resolveTerm exprs = go (8 :: Int)+ where+ go 0 _ = Nothing+ go fuel (UExpr e) = case e of+ Col nm -> case M.lookup nm exprs of+ Just ue -> go (fuel - 1) ue+ Nothing -> Just (Lin nm)+ Unary _ inner -> go (fuel - 1) (UExpr inner)+ Binary op l r+ | binaryName op == "mult" && binaryCommutative op -> do+ Lin a <- go (fuel - 1) (UExpr l)+ Lin b <- go (fuel - 1) (UExpr r)+ Just (sortProd a b)+ _ -> Nothing++-- | Products are unordered: store the pair sorted so @x*y@ and @y*x@ unify.+sortProd :: T.Text -> T.Text -> Term+sortProd a b+ | a <= b = Prod a b+ | otherwise = Prod b a++{- | From the classified roles, find the unordered pair of base columns that the+linear sums name. There must be exactly two distinct linear-sum columns.+-}+pickBaseColumns :: [(T.Text, Role)] -> Maybe (T.Text, T.Text)+pickBaseColumns roles =+ case lins of+ [a, b] | a /= b -> Just (a, b)+ _ -> Nothing+ where+ lins = M.keys (M.fromList [(c, ()) | (_, RoleLin c) <- roles])
+ src-internal/DataFrame/Internal/Aggregation/Reduction.hs view
@@ -0,0 +1,44 @@+{-# LANGUAGE ScopedTypeVariables #-}++{- | The vocabulary shared by the aggregation planner and the aggregation+kernels: the set of recognised reductions, and the admission gate that decides+whether a column may enter a Double-valued fast path.++This module holds no kernel and no policy, so the planner can depend on it+without depending on an implementation.+-}+module DataFrame.Internal.Aggregation.Reduction (+ Reduction (..),+ cleanDoubleVector,+) where++import qualified Data.Vector.Unboxed as VU++import DataFrame.Internal.Column (+ Column (..),+ materializePacked,+ )+import DataFrame.Internal.Column.Conversion (toDoubleVector)++{- | A recognised fast-path reduction over a single value column. The element+type (Int vs Double) is resolved at scatter time; sum/min/max preserve the+column's element type, everything else produces a Double column.+-}+data Reduction+ = RSum+ | RCount+ | RMin+ | RMax+ | RMean+ | RStd+ | RVar+ | RTop2Sum+ | RTop2Snd+ deriving (Eq, Show)++cleanDoubleVector :: Column -> Maybe (VU.Vector Double)+cleanDoubleVector col = case col of+ UnboxedColumn Nothing _ -> either (const Nothing) Just (toDoubleVector col)+ BoxedColumn Nothing _ -> either (const Nothing) Just (toDoubleVector col)+ p@(PackedText _ _) -> cleanDoubleVector (materializePacked p)+ _ -> Nothing
+ src-internal/DataFrame/Internal/Algorithms/Hash.hs view
@@ -0,0 +1,113 @@+{-# LANGUAGE BangPatterns #-}+{-# LANGUAGE CPP #-}+{-# LANGUAGE MagicHash #-}++{- | A poor-man's hash used by 'DataFrame.Internal.Grouping' to bucket rows+without depending on @hashable@. Each value is folded into an 'Int' with an+FxHash-style step (rotate, xor, multiply); small and not cryptographic.+-}+module DataFrame.Internal.Algorithms.Hash (+ fnvOffset,+ nullSalt,+ mixInt,+ mixDouble,+ mixBool,+ mixChar,+ mixText,+ mixBytes,+ mixShow,+) where++import Data.Bits (rotateL, unsafeShiftL, unsafeShiftR, xor)+import Data.Char (ord)+import qualified Data.Text as T+import qualified Data.Text.Array as A+#if MIN_VERSION_text(2,1,0)+import Data.Array.Byte (ByteArray (ByteArray))+#else+import Data.Text.Array (Array (ByteArray))+#endif+import Data.Text.Internal (Text (Text))+import GHC.Exts (Int (I#), indexWord8Array#, indexWord8ArrayAsWord64#)+import GHC.Word (Word64 (W64#), Word8 (W8#))++{- | FNV-1a 64-bit offset basis (used as the initial accumulator).+The literal is unsigned and exceeds 'Int' range, so we round-trip through+'Word64' to get the well-defined two's-complement bit pattern.+-}+fnvOffset :: Int+fnvOffset = fromIntegral (0xcbf29ce484222325 :: Word64)++-- | FNV-1a 64-bit prime.+fnvPrime :: Int+fnvPrime = 0x00000100000001b3++{- | Sentinel mixed in for a /null/ slot, so @Nothing@ does not hash the same as+a present value with equal bits (e.g. @Just 0@). A fixed distinctive constant+keeps null hashing deterministic; a real value equal to it collides only rarely.+-}+nullSalt :: Int+nullSalt = fromIntegral (0x9E3779B97F4A7C15 :: Word64)++{- | Mix an 'Int' into the accumulator with an FxHash-style step. The rotate+diffuses each value's bits before the next is folded in, avoiding the structured+collisions a plain xor-then-multiply produces on small/adjacent group keys.+-}+mixInt :: Int -> Int -> Int+mixInt acc x = (rotateL acc 13 `xor` x) * fnvPrime+{-# INLINE mixInt #-}++{- | Mix a 'Double' into the accumulator. Loses sub-millisecond precision+but matches the bucketing the old hashable-based code used.+-}+mixDouble :: Int -> Double -> Int+mixDouble acc d = mixInt acc (floor (d * 1000))+{-# INLINE mixDouble #-}++mixBool :: Int -> Bool -> Int+mixBool acc b = mixInt acc (if b then 1 else 0)+{-# INLINE mixBool #-}++mixChar :: Int -> Char -> Int+mixChar acc = mixInt acc . ord+{-# INLINE mixChar #-}++{- | Mix a 'T.Text' value into the accumulator over its raw UTF-8 bytes, eight at+a time. Reading a whole 'Word64' per step cuts the multiply count ~8x on long+keys while staying collision-equivalent (UTF-8 is injective).+-}+mixText :: Int -> T.Text -> Int+mixText !acc (Text arr off len) = mixBytes acc arr off len+{-# INLINE mixText #-}++{- | Mix a raw UTF-8 byte slice @[off, off+len)@ of a 'Data.Text.Array.Array'+into the accumulator, eight bytes at a time. The shared kernel behind+'mixText' and the packed-text hash path, so the two never drift.+-}+mixBytes :: Int -> A.Array -> Int -> Int -> Int+mixBytes !acc arr off len = goBytes (goWords acc off) wordsEnd+ where+ !(ByteArray ba) = arr+ !nWords = len `unsafeShiftR` 3+ !wordsEnd = off + (nWords `unsafeShiftL` 3)+ !end = off + len+ goWords !h !i+ | i >= wordsEnd = h+ | otherwise =+ let !(I# i#) = i+ !w = fromIntegral (W64# (indexWord8ArrayAsWord64# ba i#)) :: Int+ in goWords (mixInt h w) (i + 8)+ goBytes !h !i+ | i >= end = h+ | otherwise =+ let !(I# i#) = i+ !b = fromIntegral (W8# (indexWord8Array# ba i#)) :: Int+ in goBytes (mixInt h b) (i + 1)+{-# INLINE mixBytes #-}++{- | Fallback for arbitrary 'Show'-able values. Slower but covers types+without a dedicated combinator (e.g. 'Day', 'UTCTime').+-}+mixShow :: (Show a) => Int -> a -> Int+mixShow acc = mixText acc . T.pack . show+{-# INLINE mixShow #-}
+ src-internal/DataFrame/Internal/Algorithms/Rank/Radix.hs view
@@ -0,0 +1,101 @@+{-# LANGUAGE BangPatterns #-}+{-# LANGUAGE ScopedTypeVariables #-}++{- | Stable rank of a set of group representatives by ascending unsigned hash+order. Shared by the sequential and parallel group-by canonical-ordering steps+so they stay bit-for-bit identical. @O(ng)@ stable LSD radix sort.+-}+module DataFrame.Internal.Algorithms.Rank.Radix (+ rankByHash,+ sortKey,+) where++import Control.Monad (when)+import Control.Monad.Primitive (PrimMonad)+import Data.Bits (unsafeShiftR, (.&.))+import qualified Data.Vector.Unboxed as VU+import qualified Data.Vector.Unboxed.Mutable as VUM+import Data.Word (Word64)++{- | Unsigned sort key of a hash: ascending 'Word64' order of @sortKey h@ equals+ascending signed-'Int' order of @h@. Reinterpreted to 'Int' for the byte-wise+radix passes (the byte mask makes the sign extension irrelevant).+-}+sortKey :: Int -> Int+sortKey h = fromIntegral (fromIntegral h + 0x8000000000000000 :: Word64)+{-# INLINE sortKey #-}++-- | See the module header. @readHash@ supplies the hash of local group @gid@.+rankByHash ::+ forall m. (PrimMonad m) => (Int -> m Int) -> Int -> m (VU.Vector Int)+rankByHash readHash ng = do+ rankM <- VUM.new (max 1 ng)+ if ng <= 1+ then when (ng == 1) (VUM.unsafeWrite rankM 0 0)+ else do+ keysA <- VUM.new ng+ orderA <- VUM.new ng+ let seed !i+ | i >= ng = pure ()+ | otherwise = do+ h <- readHash i+ VUM.unsafeWrite keysA i (sortKey h)+ VUM.unsafeWrite orderA i i+ seed (i + 1)+ seed 0+ keysB <- VUM.new ng+ orderB <- VUM.new ng+ counts <- VUM.new 256+ let pass ::+ Int ->+ VUM.MVector (VUM.PrimState m) Int ->+ VUM.MVector (VUM.PrimState m) Int ->+ VUM.MVector (VUM.PrimState m) Int ->+ VUM.MVector (VUM.PrimState m) Int ->+ m ()+ pass !shiftBits !srcK !srcO !dstK !dstO = do+ VUM.set counts 0+ let count !i+ | i >= ng = pure ()+ | otherwise = do+ k <- VUM.unsafeRead srcK i+ let !b = (k `unsafeShiftR` shiftBits) .&. 0xff+ VUM.unsafeRead counts b >>= VUM.unsafeWrite counts b . (+ 1)+ count (i + 1)+ count 0+ let scan !b !acc+ | b >= 256 = pure ()+ | otherwise = do+ c <- VUM.unsafeRead counts b+ VUM.unsafeWrite counts b acc+ scan (b + 1) (acc + c)+ scan 0 0+ let place !i+ | i >= ng = pure ()+ | otherwise = do+ k <- VUM.unsafeRead srcK i+ o <- VUM.unsafeRead srcO i+ let !b = (k `unsafeShiftR` shiftBits) .&. 0xff+ pos <- VUM.unsafeRead counts b+ VUM.unsafeWrite counts b (pos + 1)+ VUM.unsafeWrite dstK pos k+ VUM.unsafeWrite dstO pos o+ place (i + 1)+ place 0+ pass 0 keysA orderA keysB orderB+ pass 8 keysB orderB keysA orderA+ pass 16 keysA orderA keysB orderB+ pass 24 keysB orderB keysA orderA+ pass 32 keysA orderA keysB orderB+ pass 40 keysB orderB keysA orderA+ pass 48 keysA orderA keysB orderB+ pass 56 keysB orderB keysA orderA+ let inv !r+ | r >= ng = pure ()+ | otherwise = do+ g <- VUM.unsafeRead orderA r+ VUM.unsafeWrite rankM g r+ inv (r + 1)+ inv 0+ VU.unsafeFreeze rankM+{-# INLINEABLE rankByHash #-}
+ src-internal/DataFrame/Internal/Algorithms/Sort/Radix/Parallel.hs view
@@ -0,0 +1,332 @@+{-# LANGUAGE BangPatterns #-}+{-# LANGUAGE ScopedTypeVariables #-}++{- | Parallel stable sort of row indices by ascending unsigned order of a per-row+'Int' hash, used by the join build side. A counting sort buckets rows into+key-ordered partitions that workers LSD-radix-sort in parallel, with no merge step.+-}+module DataFrame.Internal.Algorithms.Sort.Radix.Parallel (+ parSortByHash,+ parSortThreshold,+) where++import Control.Concurrent (getNumCapabilities)+import Control.Monad (forM_, when)+import Data.Bits (countLeadingZeros, unsafeShiftR, (.&.))+import qualified Data.Vector.Unboxed as VU+import qualified Data.Vector.Unboxed.Mutable as VUM+import Data.Word (Word64)+import DataFrame.Internal.Algorithms.Rank.Radix (sortKey)+import DataFrame.Internal.Control.Concurrent (+ capabilities,+ forkJoin,+ forkJoin_,+ pooledIndices,+ )+import System.IO.Unsafe (unsafePerformIO)++{- | Below this many rows the partition/fork overhead is not worth it; the+caller's sequential LSD radix path is used instead.+-}+parSortThreshold :: Int+parSortThreshold = 500000++{- | Top-bits partition index of a hash: the high @64 - shift@ bits of its+unsigned 'sortKey'. Ascending partition order equals ascending key order.+-}+partIx :: Int -> Int -> Int+partIx shift h = fromIntegral ((fromIntegral (sortKey h) :: Word64) `unsafeShiftR` shift)+{-# INLINE partIx #-}++-- | Number of partitions: a power of two, at least @4 * caps@, floored at 256.+numPartitionsFor :: Int -> Int+numPartitionsFor caps = go 1+ where+ target = max 256 (4 * caps)+ go p+ | p >= target = p+ | otherwise = go (p * 2)++-- | @floor (log2 x)@ for a power-of-two @x@.+intLog2 :: Int -> Int+intLog2 x = 63 - countLeadingZeros x+{-# INLINE intLog2 #-}++{- | Parallel stable sort of @[0, n)@ by ascending unsigned hash order. See the+module header for the ordering contract.+-}+parSortByHash :: Int -> VU.Vector Int -> (VU.Vector Int, VU.Vector Int)+parSortByHash n hashes+ | n <= 1 =+ (hashes, VU.enumFromN 0 n)+ | n < parSortThreshold || capabilities <= 1 =+ seqSortByHash n hashes+ | otherwise = unsafePerformIO (parSortByHashIO n hashes)+{-# NOINLINE parSortByHash #-}++-------------------------------------------------------------------------------+-- Sequential LSD radix sort (also the per-partition worker kernel)+-------------------------------------------------------------------------------++{- | Stable LSD radix sort of @[0, n)@ by ascending 'sortKey' of their hash, 8+bits per pass over the full 64-bit key. Returns @(sortedHashes, sortedIndices)@.+-}+seqSortByHash :: Int -> VU.Vector Int -> (VU.Vector Int, VU.Vector Int)+seqSortByHash n hashes = unsafePerformIO $ do+ keysA <- VUM.new n+ orderA <- VUM.new n+ let seed !i+ | i >= n = pure ()+ | otherwise = do+ VUM.unsafeWrite keysA i (sortKey (VU.unsafeIndex hashes i))+ VUM.unsafeWrite orderA i i+ seed (i + 1)+ seed 0+ keysB <- VUM.new n+ orderB <- VUM.new n+ radixPasses n keysA orderA keysB orderB+ order <- VU.unsafeFreeze orderA+ pure (VU.unsafeBackpermute hashes order, order)++{- | Run all eight stable 8-bit LSD passes, ping-ponging between the two+key/order buffer pairs so the sorted order lands back in @(keysA, orderA)@.+@keysA[i]@ must already hold @sortKey (hash of orderA[i])@ on entry.+-}+radixPasses ::+ Int ->+ VUM.IOVector Int ->+ VUM.IOVector Int ->+ VUM.IOVector Int ->+ VUM.IOVector Int ->+ IO ()+radixPasses = radixPassesN 8++{- | Run the first @np@ stable 8-bit LSD passes (bits @0 .. 8*np-1@),+ping-ponging between the buffer pairs. For odd @np@ the sorted order lands in+@(keysB, orderB)@, for even in @(keysA, orderA)@. Callers whose rows share+their top bytes (per-partition sorts partitioned on the top byte) can pass+@np = 7@: the eighth pass would be a stable identity copy.+-}+radixPassesN ::+ Int ->+ Int ->+ VUM.IOVector Int ->+ VUM.IOVector Int ->+ VUM.IOVector Int ->+ VUM.IOVector Int ->+ IO ()+radixPassesN np n keysA orderA keysB orderB = do+ counts <- VUM.new 256+ let pass ::+ Int ->+ VUM.IOVector Int ->+ VUM.IOVector Int ->+ VUM.IOVector Int ->+ VUM.IOVector Int ->+ IO ()+ pass !shiftBits !srcK !srcO !dstK !dstO = do+ VUM.set counts 0+ let count !i+ | i >= n = pure ()+ | otherwise = do+ k <- VUM.unsafeRead srcK i+ let !b = (k `unsafeShiftR` shiftBits) .&. 0xff+ VUM.unsafeRead counts b >>= VUM.unsafeWrite counts b . (+ 1)+ count (i + 1)+ count 0+ let scan !b !acc+ | b >= 256 = pure ()+ | otherwise = do+ c <- VUM.unsafeRead counts b+ VUM.unsafeWrite counts b acc+ scan (b + 1) (acc + c)+ scan 0 0+ let place !i+ | i >= n = pure ()+ | otherwise = do+ k <- VUM.unsafeRead srcK i+ o <- VUM.unsafeRead srcO i+ let !b = (k `unsafeShiftR` shiftBits) .&. 0xff+ pos <- VUM.unsafeRead counts b+ VUM.unsafeWrite counts b (pos + 1)+ VUM.unsafeWrite dstK pos k+ VUM.unsafeWrite dstO pos o+ place (i + 1)+ place 0+ run !k+ | k >= np = pure ()+ | even k = pass (8 * k) keysA orderA keysB orderB >> run (k + 1)+ | otherwise = pass (8 * k) keysB orderB keysA orderA >> run (k + 1)+ run 0++-------------------------------------------------------------------------------+-- Parallel path: counting-sort partition, then per-partition sort in parallel+-------------------------------------------------------------------------------++parSortByHashIO :: Int -> VU.Vector Int -> IO (VU.Vector Int, VU.Vector Int)+parSortByHashIO n hashes = do+ caps <- getNumCapabilities+ let !p = numPartitionsFor caps+ !shift = 64 - intLog2 p+ (partStart, partRows, partHashes) <- partitionRows n hashes p shift+ outOrder <- VUM.new n+ outKeys <- VUM.new n+ sortPartitions caps p partStart partRows partHashes outOrder outKeys+ order <- VU.unsafeFreeze outOrder+ sortedHashes <- VU.unsafeFreeze outKeys+ pure (sortedHashes, order)++{- | Bucket every row index into its top-bits partition by a counting sort.+Returns the exclusive prefix sum @partStart@ (length @p+1@, @partStart[p] == n@),+the row indices laid out partition-by-partition in ascending key order, and+each sorted position's hash in the same layout (so downstream passes read+hashes sequentially instead of a random @hashes[row]@ gather per row).++Runs chunked across capabilities: per-chunk partition histograms are prefix+summed (in chunk order) into disjoint per-chunk write cursors, so the scatter+threads never contend and each partition keeps its rows in ascending original+row order — bit-for-bit the sequential counting sort's layout.+-}+partitionRows ::+ Int ->+ VU.Vector Int ->+ Int ->+ Int ->+ IO (VU.Vector Int, VU.Vector Int, VU.Vector Int)+partitionRows n hashes p shift = do+ caps <- getNumCapabilities+ let chunks = rowChunks caps n+ cursors <- forkJoin [histChunk hashes p shift lo hi | (lo, hi) <- chunks]+ -- Exclusive prefix over partitions (outer) and chunks (inner): partStart+ -- from the totals, and each chunk's histogram rewritten into its cursor.+ partStartM <- VUM.new (p + 1)+ let seed !pp !acc+ | pp >= p = VUM.unsafeWrite partStartM p acc+ | otherwise = do+ VUM.unsafeWrite partStartM pp acc+ let inner [] !a = pure a+ inner (cur : rest) !a = do+ t <- VUM.unsafeRead cur pp+ VUM.unsafeWrite cur pp a+ inner rest (a + t)+ acc' <- inner cursors acc+ seed (pp + 1) acc'+ seed 0 0+ rowsM <- VUM.new (max 1 n)+ rowHashM <- VUM.new (max 1 n)+ forkJoin_+ [ scatterChunk hashes shift cur rowsM rowHashM lo hi+ | ((lo, hi), cur) <- zip chunks cursors+ ]+ partStart <- VU.unsafeFreeze partStartM+ partRows <- VU.unsafeFreeze rowsM+ partHashes <- VU.unsafeFreeze rowHashM+ pure (partStart, partRows, partHashes)++-- | Contiguous near-equal row chunks, one per capability; empties dropped.+rowChunks :: Int -> Int -> [(Int, Int)]+rowChunks caps n =+ [ (lo, hi)+ | w <- [0 .. caps - 1]+ , let lo = min n (w * per)+ , let hi = min n (lo + per)+ , lo < hi+ ]+ where+ !per = (n + max 1 caps - 1) `div` max 1 caps++-- | Per-partition counts of one row chunk.+histChunk :: VU.Vector Int -> Int -> Int -> Int -> Int -> IO (VUM.IOVector Int)+histChunk hashes p shift lo hi = do+ acc <- VUM.replicate p (0 :: Int)+ let go !i+ | i >= hi = pure acc+ | otherwise = do+ let !pp = partIx shift (VU.unsafeIndex hashes i)+ c <- VUM.unsafeRead acc pp+ VUM.unsafeWrite acc pp (c + 1)+ go (i + 1)+ go lo++{- | Scatter one row chunk into the partitioned layout using the chunk's+pre-summed cursor (disjoint write regions per chunk, no contention).+-}+scatterChunk ::+ VU.Vector Int ->+ Int ->+ VUM.IOVector Int ->+ VUM.IOVector Int ->+ VUM.IOVector Int ->+ Int ->+ Int ->+ IO ()+scatterChunk hashes shift cursor rowsM rowHashM lo hi = go lo+ where+ go !i+ | i >= hi = pure ()+ | otherwise = do+ let !h = VU.unsafeIndex hashes i+ !pp = partIx shift h+ pos <- VUM.unsafeRead cursor pp+ VUM.unsafeWrite rowsM pos i+ VUM.unsafeWrite rowHashM pos h+ VUM.unsafeWrite cursor pp (pos + 1)+ go (i + 1)++{- | Stable-sort each partition by full key, writing sorted original indices+into @outOrder@ and their hashes into @outKeys@ at the partition's slot range.+Forks @caps@ workers that pull partition indices off a shared atomic counter.+Within a partition the counting sort already left rows in ascending original+order, so the LSD radix sort's stability reproduces the global @(key, row)@+order. Partitions below two elements are already sorted (counting sort kept+original order) and are copied directly.++@partHashes@ is the partition-layout hash vector from 'partitionRows', so+seeding reads hashes sequentially; only 7 LSD passes run (the top byte is the+partition byte, constant within a partition), and the sorted hash is recovered+from the sort key ('sortKey' is self-inverse) instead of a random gather.+-}+sortPartitions ::+ Int ->+ Int ->+ VU.Vector Int ->+ VU.Vector Int ->+ VU.Vector Int ->+ VUM.IOVector Int ->+ VUM.IOVector Int ->+ IO ()+sortPartitions caps p partStart partRows partHashes outOrder outKeys =+ pooledIndices caps p sortOne+ where+ sortOne !pp = do+ let !s = VU.unsafeIndex partStart pp+ !e = VU.unsafeIndex partStart (pp + 1)+ !sz = e - s+ when (sz > 0) $+ if sz == 1+ then do+ VUM.unsafeWrite outOrder s (VU.unsafeIndex partRows s)+ VUM.unsafeWrite outKeys s (VU.unsafeIndex partHashes s)+ else do+ keysA <- VUM.new sz+ orderA <- VUM.new sz+ let seed !i+ | i >= sz = pure ()+ | otherwise = do+ VUM.unsafeWrite keysA i (sortKey (VU.unsafeIndex partHashes (s + i)))+ VUM.unsafeWrite orderA i (VU.unsafeIndex partRows (s + i))+ seed (i + 1)+ seed 0+ keysB <- VUM.new sz+ orderB <- VUM.new sz+ radixPassesN 7 sz keysA orderA keysB orderB+ let emit !i+ | i >= sz = pure ()+ | otherwise = do+ o <- VUM.unsafeRead orderB i+ k <- VUM.unsafeRead keysB i+ VUM.unsafeWrite outOrder (s + i) o+ -- sortKey is an involution: recover the hash.+ VUM.unsafeWrite outKeys (s + i) (sortKey k)+ emit (i + 1)+ emit 0
src-internal/DataFrame/Internal/Column.hs view
@@ -1,1835 +1,26 @@-{-# LANGUAGE AllowAmbiguousTypes #-}-{-# LANGUAGE BangPatterns #-}-{-# LANGUAGE ConstraintKinds #-}-{-# LANGUAGE DataKinds #-}-{-# LANGUAGE FlexibleContexts #-}-{-# LANGUAGE FlexibleInstances #-}-{-# LANGUAGE GADTs #-}-{-# LANGUAGE InstanceSigs #-}-{-# LANGUAGE LambdaCase #-}-{-# LANGUAGE MultiParamTypeClasses #-}-{-# LANGUAGE OverloadedStrings #-}-{-# LANGUAGE PolyKinds #-}-{-# LANGUAGE RankNTypes #-}-{-# LANGUAGE ScopedTypeVariables #-}-{-# LANGUAGE TypeApplications #-}-{-# LANGUAGE TypeFamilies #-}-{-# LANGUAGE UndecidableInstances #-}--module DataFrame.Internal.Column where--import qualified Data.Text as T-import qualified Data.Vector as VB-import qualified Data.Vector.Generic as VG-import qualified Data.Vector.Mutable as VBM-import qualified Data.Vector.Unboxed as VU-import qualified Data.Vector.Unboxed.Mutable as VUM--import Control.Exception (throw)-import Control.Monad (forM_, when)-import Control.Monad.ST (ST, runST)-import Data.Bits (- complement,- popCount,- setBit,- shiftL,- shiftR,- testBit,- (.&.),- )-import Data.Kind (Type)-import Data.Maybe-import Data.Type.Equality (TestEquality (..))-import Data.Word (Word8)-import DataFrame.Errors-import DataFrame.Internal.PackedText (- PackedTextData (..),- packedGather,- packedIndexText,- packedLength,- packedSlice,- packedTake,- sliceEqBytes,- )-import DataFrame.Internal.Types-import System.IO.Unsafe (unsafePerformIO)-import System.Random-import Type.Reflection---- | A bit-packed validity bitmap. Bit @i@ = 1 means row @i@ is valid (not null).-type Bitmap = VU.Vector Word8--{- | Type-erased column GADT. Pattern-matching on the constructor recovers the-representation; nullability is an optional bit-packed 'Bitmap' (@Nothing@ = no-nulls, @Just bm@ = bit @i@ set iff row @i@ is valid).--}-data Column where- BoxedColumn :: (Columnable a) => Maybe Bitmap -> VB.Vector a -> Column- UnboxedColumn ::- (Columnable a, VU.Unbox a) => Maybe Bitmap -> VU.Vector a -> Column- -- Efficient intermediate formats.- -- Bit-packed Text: shared UTF-8 byte buffer + row offsets + optional bitmap;- -- Text is materialized on demand. Only CSV ingest emits this; user-built- -- Text columns stay 'BoxedColumn'.- PackedText :: Maybe Bitmap -> {-# UNPACK #-} !PackedTextData -> Column- -- A join's same-named non-key column pair ('mergeColumns'): both sides- -- keep their native (packed/dict/unboxed) representation; per-row 'These'- -- values only materialize on element access ('materializeMerged').- MergedColumn :: !Column -> !Column -> Column--{- | A mutable companion struct to dataframe columns.--Used mostly as an intermediate structure for I/O.--}-data MutableColumn where- MBoxedColumn :: (Columnable a) => VBM.IOVector a -> MutableColumn- MUnboxedColumn :: (Columnable a, VU.Unbox a) => VUM.IOVector a -> MutableColumn---- ------------------------------------------------------------------------------ Bitmap helpers--- ------------------------------------------------------------------------------- | Test whether row @i@ is valid (not null) in a bitmap.-bitmapTestBit :: Bitmap -> Int -> Bool-bitmapTestBit bm i = testBit (VU.unsafeIndex bm (i `shiftR` 3)) (i .&. 7)-{-# INLINE bitmapTestBit #-}---- | Build a fully-valid bitmap for @n@ rows (all bits set).-allValidBitmap :: Int -> Bitmap-allValidBitmap n =- let bytes = (n + 7) `shiftR` 3- lastBits = n .&. 7- full = VU.replicate (bytes - 1) 0xFF- lastByte = if lastBits == 0 then 0xFF else (1 `shiftL` lastBits) - 1- in if bytes == 0 then VU.empty else VU.snoc full lastByte-{-# INLINE allValidBitmap #-}--{- | Build a bitmap from a @VU.Vector Word8@ validity vector-(1 = valid, 0 = null), as produced by Arrow / Parquet decoders.--}-buildBitmapFromValid :: VU.Vector Word8 -> Bitmap-buildBitmapFromValid valid =- let n = VU.length valid- bytes = (n + 7) `shiftR` 3- in VU.generate bytes $ \b ->- let base = b `shiftL` 3- setBitIf acc bit =- let idx = base + bit- in if idx < n && VU.unsafeIndex valid idx /= 0- then setBit acc bit- else acc- in foldl setBitIf (0 :: Word8) [0 .. 7]--{- | Build a bitmap from a list of null-row indices.-@nullIdxs@ are the positions that are NULL.--}-buildBitmapFromNulls :: Int -> [Int] -> Bitmap-buildBitmapFromNulls n nullIdxs =- let base = allValidBitmap n- in VU.modify- ( \mv ->- forM_ nullIdxs $ \i -> do- let byteIdx = i `shiftR` 3- bitIdx = i .&. 7- v <- VUM.unsafeRead mv byteIdx- VUM.unsafeWrite mv byteIdx (clearBit8 v bitIdx)- )- base- where- clearBit8 :: Word8 -> Int -> Word8- clearBit8 b bit = b .&. complement (1 `shiftL` bit)---- | Slice a bitmap for rows @[start .. start+len-1]@.-bitmapSlice :: Int -> Int -> Bitmap -> Bitmap-bitmapSlice start len bm- | start .&. 7 == 0 =- let startByte = start `shiftR` 3- bytes = min ((len + 7) `shiftR` 3) (VU.length bm - startByte)- in VU.slice startByte bytes bm- | otherwise =- let n = min len (VU.length bm `shiftL` 3 - start)- in buildBitmapFromValid $- VU.generate n $- \i -> if bitmapTestBit bm (start + i) then 1 else 0---- | Concatenate two bitmaps covering @n1@ and @n2@ rows respectively.-bitmapConcat :: Int -> Bitmap -> Int -> Bitmap -> Bitmap-bitmapConcat n1 bm1 n2 bm2 =- buildBitmapFromValid $- VU.generate (n1 + n2) $ \i ->- if i < n1- then if bitmapTestBit bm1 i then 1 else 0- else if bitmapTestBit bm2 (i - n1) then 1 else 0---- | Combine two bitmaps with AND (both must be valid for result to be valid).-mergeBitmaps :: Bitmap -> Bitmap -> Bitmap-mergeBitmaps = VU.zipWith (.&.)--{- | Materialize a nullable column from @VB.Vector (Maybe a)@; picks 'UnboxedColumn'-when @a@ is unboxable, else 'BoxedColumn'. Always attaches a bitmap so the column-reads as nullable even with no 'Nothing' values.--}-fromMaybeVec :: forall a. (Columnable a) => VB.Vector (Maybe a) -> Column-fromMaybeVec v = case sUnbox @a of- STrue -> fromMaybeVecUnboxed v- SFalse ->- let n = VB.length v- nullIdxs = [i | i <- [0 .. n - 1], isNothing (VB.unsafeIndex v i)]- bm = if null nullIdxs then allValidBitmap n else buildBitmapFromNulls n nullIdxs- dat = VB.map (fromMaybe (errorWithoutStackTrace "fromMaybeVec: Nothing slot")) v- in BoxedColumn (Just bm) dat--{- | Materialize a nullable 'UnboxedColumn' to @VB.Vector (Maybe a)@ using runST.-Always attaches a bitmap so the column is recognized as nullable even when-no 'Nothing' values are present (preserves the Maybe type marker).--}-fromMaybeVecUnboxed ::- forall a. (Columnable a, VU.Unbox a) => VB.Vector (Maybe a) -> Column-fromMaybeVecUnboxed v =- let n = VB.length v- nullIdxs = [i | i <- [0 .. n - 1], isNothing (VB.unsafeIndex v i)]- bm = if null nullIdxs then allValidBitmap n else buildBitmapFromNulls n nullIdxs- dat = runST $ do- mv <- VUM.new n- VG.iforM_ v $ \i mx -> forM_ mx (VUM.unsafeWrite mv i)- VU.unsafeFreeze mv- in UnboxedColumn (Just bm) dat---- | Whether row @i@ is null, respecting the bitmap.-columnElemIsNull :: Column -> Int -> Bool-columnElemIsNull (BoxedColumn (Just bm) _) i = not (bitmapTestBit bm i)-columnElemIsNull (UnboxedColumn (Just bm) _) i = not (bitmapTestBit bm i)-columnElemIsNull (PackedText (Just bm) _) i = not (bitmapTestBit bm i)-columnElemIsNull _ _ = False---- | Return the 'Maybe Bitmap' from a column.-columnBitmap :: Column -> Maybe Bitmap-columnBitmap (BoxedColumn bm _) = bm-columnBitmap (UnboxedColumn bm _) = bm-columnBitmap (PackedText bm _) = bm-columnBitmap (MergedColumn _ _) = Nothing--{- | Decode a 'PackedText' into a @BoxedColumn Text@ (bit-identical to-materializing at freeze). Identity on every other column.--}-materializePacked :: Column -> Column-materializePacked (PackedText bm p) =- BoxedColumn bm (VB.generate (packedLength p) (packedIndexText p))-materializePacked c = c-{-# INLINE materializePacked #-}---- | Whether a column is a 'PackedText'.-isPackedText :: Column -> Bool-isPackedText (PackedText _ _) = True-isPackedText _ = False-{-# INLINE isPackedText #-}---- | Whether a column is a 'MergedColumn'.-isMergedColumn :: Column -> Bool-isMergedColumn (MergedColumn _ _) = True-isMergedColumn _ = False-{-# INLINE isMergedColumn #-}--{- | 'MergedColumn' defers element construction, so forcing must still surface-the one deferred error — a row null on both sides — inside strict IO/executor-boundaries. O(rows) bitmap walk, no allocation; both-null needs a bitmap on-each side, so anything else passes immediately.--}-checkMergedNoBothNull :: Column -> Column -> ()-checkMergedNoBothNull a b = case (columnBitmap a, columnBitmap b) of- (Just ba, Just bb) ->- let !n = min (columnLength a) (columnLength b)- go !i- | i >= n = ()- | bitmapTestBit ba i || bitmapTestBit bb i = go (i + 1)- | otherwise = error "mergeColumns: both null"- in go 0- _ -> ()---- ------------------------------------------------------------------------------ End bitmap helpers--- -----------------------------------------------------------------------------{- | A wrapper around the type-erased 'Column' carrying a phantom element type,-used to type-check expressions. The phantom is not guaranteed to match the-underlying vector's type.--}-data TypedColumn a where- TColumn :: (Columnable a) => Column -> TypedColumn a--instance (Eq a) => Eq (TypedColumn a) where- (==) :: (Eq a) => TypedColumn a -> TypedColumn a -> Bool- (==) (TColumn a) (TColumn b) = a == b---- | Gets the underlying value from a TypedColumn.-unwrapTypedColumn :: TypedColumn a -> Column-unwrapTypedColumn (TColumn value) = value---- | Gets the underlying vector from a TypedColumn.-vectorFromTypedColumn :: TypedColumn a -> VB.Vector a-vectorFromTypedColumn (TColumn value) = either throw id (toVector value)---- | Checks if a column contains missing values (has a bitmap).-hasMissing :: Column -> Bool-hasMissing (BoxedColumn (Just _) _) = True-hasMissing (UnboxedColumn (Just _) _) = True-hasMissing (PackedText (Just _) _) = True-hasMissing _ = False---- | Checks if a column contains only missing values.-allMissing :: Column -> Bool-allMissing (BoxedColumn (Just bm) col) = VU.all (== 0) bm && not (VB.null col)-allMissing (UnboxedColumn (Just bm) col) = VU.all (== 0) bm && not (VU.null col)-allMissing (PackedText (Just bm) p) = VU.all (== 0) bm && packedLength p > 0-allMissing _ = False---- | Checks if a column contains numeric values.-isNumeric :: Column -> Bool-isNumeric c@(MergedColumn _ _) = isNumeric (mergedHead c)-isNumeric (UnboxedColumn _ (_vec :: VU.Vector a)) = case sNumeric @a of- STrue -> True- _ -> False-isNumeric (BoxedColumn _ (_vec :: VB.Vector a)) = case testEquality (typeRep @a) (typeRep @Integer) of- Nothing -> False- Just Refl -> True-isNumeric (PackedText _ _) = False--{- | Whether the column stores element type @a@. For nullable columns, also-'True' when @a = Maybe b@ and the column stores @b@ internally.--}-hasElemType :: forall a. (Columnable a) => Column -> Bool-hasElemType = \case- BoxedColumn bm (_column :: VB.Vector b) -> checkBoxed bm (typeRep @b)- UnboxedColumn bm (_column :: VU.Vector b) -> checkUnboxed bm (typeRep @b)- PackedText bm _ -> checkBoxed bm (typeRep @T.Text)- c@(MergedColumn _ _) -> hasElemType @a (mergedHead c)- where- directMatch :: forall (b :: Type). TypeRep b -> Bool- directMatch = isJust . testEquality (typeRep @a)- checkMaybe :: forall (b :: Type). TypeRep b -> Bool- checkMaybe tb = case typeRep @a of- App tMaybe tInner -> case eqTypeRep tMaybe (typeRep @Maybe) of- Just HRefl -> isJust (testEquality tInner tb)- Nothing -> False- _ -> False- checkBoxed :: forall (b :: Type). Maybe Bitmap -> TypeRep b -> Bool- checkBoxed bm tb = directMatch tb || (isJust bm && checkMaybe tb)- checkUnboxed :: forall (b :: Type). Maybe Bitmap -> TypeRep b -> Bool- checkUnboxed bm tb = directMatch tb || (isJust bm && checkMaybe tb)---- | An internal/debugging function to get the column type of a column.-columnVersionString :: Column -> String-columnVersionString column = case column of- BoxedColumn Nothing _ -> "Boxed"- BoxedColumn (Just _) _ -> "NullableBoxed"- UnboxedColumn Nothing _ -> "Unboxed"- UnboxedColumn (Just _) _ -> "NullableUnboxed"- PackedText Nothing _ -> "Boxed"- PackedText (Just _) _ -> "NullableBoxed"- MergedColumn _ _ -> columnVersionString (mergedHead column)--{- | An internal/debugging function to get the type stored in the outermost vector-of a column.--}-columnTypeString :: Column -> String-columnTypeString column = case column of- BoxedColumn Nothing (_ :: VB.Vector a) -> show (typeRep @a)- BoxedColumn (Just _) (_ :: VB.Vector a) -> showMaybeType @a- UnboxedColumn Nothing (_ :: VU.Vector a) -> show (typeRep @a)- UnboxedColumn (Just _) (_ :: VU.Vector a) -> showMaybeType @a- PackedText Nothing _ -> show (typeRep @T.Text)- PackedText (Just _) _ -> showMaybeType @T.Text- MergedColumn _ _ -> columnTypeString (mergedHead column)- where- showMaybeType :: forall a. (Typeable a) => String- showMaybeType =- let s = show (typeRep @a)- in "Maybe " ++ if ' ' `elem` s then "(" ++ s ++ ")" else s--instance (Show a) => Show (TypedColumn a) where- show :: (Show a) => TypedColumn a -> String- show (TColumn col) = show col--{- | Force evaluation of all elements in a column. Replacement for the removed-@instance NFData Column@; used by the IO and lazy-executor strict paths.--}-forceColumn :: Column -> ()-forceColumn (BoxedColumn Nothing (v :: VB.Vector a)) = VB.foldl' (const (`seq` ())) () v-forceColumn (BoxedColumn (Just bm) (v :: VB.Vector a)) =- let n = VB.length v- go !i- | i >= n = ()- | bitmapTestBit bm i = VB.unsafeIndex v i `seq` go (i + 1)- | otherwise = go (i + 1)- in go 0-forceColumn (UnboxedColumn _ v) = v `seq` ()-forceColumn (PackedText _ (PackedTextData arr offs sel _)) = arr `seq` offs `seq` sel `seq` ()-forceColumn (MergedColumn a b) =- forceColumn a `seq` forceColumn b `seq` checkMergedNoBothNull a b--instance Show Column where- show :: Column -> String- show c@(MergedColumn _ _) = show (materializeMerged c)- show (BoxedColumn Nothing column) = show column- show (BoxedColumn (Just bm) column) =- let n = VB.length column- elems =- [ if bitmapTestBit bm i then show (VB.unsafeIndex column i) else "null"- | i <- [0 .. n - 1]- ]- in "[" ++ foldl (\acc e -> if null acc then e else acc ++ "," ++ e) "" elems ++ "]"- show (UnboxedColumn Nothing column) = show column- show (UnboxedColumn (Just bm) column) =- let n = VU.length column- elems =- [ if bitmapTestBit bm i then show (VU.unsafeIndex column i) else "null"- | i <- [0 .. n - 1]- ]- in "[" ++ foldl (\acc e -> if null acc then e else acc ++ "," ++ e) "" elems ++ "]"- show c@(PackedText _ _) = show (materializePacked c)--{- | Compare two nullable boxed columns element by element, skipping null slots.-Uses a manual loop to avoid stream fusion forcing null-slot error thunks.--}-eqBoxedCols ::- (Eq a) => Maybe Bitmap -> VB.Vector a -> Maybe Bitmap -> VB.Vector a -> Bool-eqBoxedCols bm1 a bm2 b- | VB.length a /= VB.length b = False- | otherwise = go 0- where- !n = VB.length a- go !i- | i >= n = True- | nullA || nullB = (nullA == nullB) && go (i + 1)- | VB.unsafeIndex a i == VB.unsafeIndex b i = go (i + 1)- | otherwise = False- where- nullA = maybe False (\bm -> not (bitmapTestBit bm i)) bm1- nullB = maybe False (\bm -> not (bitmapTestBit bm i)) bm2-{-# INLINE eqBoxedCols #-}--instance Eq Column where- (==) :: Column -> Column -> Bool- (==) (BoxedColumn bm1 (a :: VB.Vector t1)) (BoxedColumn bm2 (b :: VB.Vector t2)) =- case testEquality (typeRep @t1) (typeRep @t2) of- Nothing -> False- Just Refl -> eqBoxedCols bm1 a bm2 b- (==) (UnboxedColumn bm1 (a :: VU.Vector t1)) (UnboxedColumn bm2 (b :: VU.Vector t2)) =- case testEquality (typeRep @t1) (typeRep @t2) of- Nothing -> False- Just Refl ->- VU.length a == VU.length b- && VU.and- ( VU.imap- ( \i x ->- let nullA = maybe False (\bm -> not (bitmapTestBit bm i)) bm1- nullB = maybe False (\bm -> not (bitmapTestBit bm i)) bm2- in if nullA || nullB then nullA == nullB else x == VU.unsafeIndex b i- )- a- )- (==) lhs@(MergedColumn _ _) rhs = materializeMerged lhs == rhs- (==) lhs rhs@(MergedColumn _ _) = lhs == materializeMerged rhs- (==) (PackedText bm1 p1) (PackedText bm2 p2) = eqPackedCols bm1 p1 bm2 p2- (==) lhs@(PackedText _ _) rhs = materializePacked lhs == rhs- (==) lhs rhs@(PackedText _ _) = lhs == materializePacked rhs- (==) _ _ = False--{- | Byte-slice equality of two packed-text columns, skipping null slots-(a null compares equal only to a null), mirroring 'eqBoxedCols'.--}-eqPackedCols ::- Maybe Bitmap -> PackedTextData -> Maybe Bitmap -> PackedTextData -> Bool-eqPackedCols bm1 p1 bm2 p2- | packedLength p1 /= packedLength p2 = False- | otherwise = go 0- where- !n = packedLength p1- go !i- | i >= n = True- | nullA || nullB = (nullA == nullB) && go (i + 1)- | otherwise =- let (a1, o1, l1) = packedSlice p1 i- (a2, o2, l2) = packedSlice p2 i- in sliceEqBytes a1 o1 l1 a2 o2 l2 && go (i + 1)- where- nullA = maybe False (\bm -> not (bitmapTestBit bm i)) bm1- nullB = maybe False (\bm -> not (bitmapTestBit bm i)) bm2-{-# INLINE eqPackedCols #-}--{- | A class for converting a vector to a column of the appropriate type.-Given each Rep we tell the `toColumnRep` function which Column type to pick.--}-class ColumnifyRep (r :: Rep) a where- toColumnRep :: VB.Vector a -> Column---- | Constraint synonym for what we can put into columns.-type Columnable a =- ( Columnable' a- , ColumnifyRep (KindOf a) a- , UnboxIf a- , IntegralIf a- , FloatingIf a- , SBoolI (Unboxable a)- , SBoolI (Numeric a)- , SBoolI (IntegralTypes a)- , SBoolI (FloatingTypes a)- )--instance- (Columnable a, VU.Unbox a) =>- ColumnifyRep 'RUnboxed a- where- toColumnRep :: (Columnable a, VUM.Unbox a) => VB.Vector a -> Column- toColumnRep v = UnboxedColumn Nothing (VU.convert v)--instance- (Columnable a) =>- ColumnifyRep 'RBoxed a- where- toColumnRep :: (Columnable a) => VB.Vector a -> Column- toColumnRep = BoxedColumn Nothing--instance- (Columnable a) =>- ColumnifyRep 'RNullableBoxed (Maybe a)- where- toColumnRep :: (Columnable a) => VB.Vector (Maybe a) -> Column- toColumnRep = fromMaybeVec--{- | O(n) Convert a vector to a column. Automatically picks the best representation of a vector to store the underlying data in.--__Examples:__--@-> import qualified Data.Vector as V-> fromVector (VB.fromList [(1 :: Int), 2, 3, 4])-[1,2,3,4]-@--}-fromVector ::- forall a.- (Columnable a, ColumnifyRep (KindOf a) a) =>- VB.Vector a -> Column-fromVector = toColumnRep @(KindOf a)--{- | O(n) Convert an unboxed vector to a column. This avoids the extra conversion if you already have the data in an unboxed vector.--__Examples:__--@-> import qualified Data.Vector.Unboxed as V-> fromUnboxedVector (VB.fromList [(1 :: Int), 2, 3, 4])-[1,2,3,4]-@--}-fromUnboxedVector ::- forall a. (Columnable a, VU.Unbox a) => VU.Vector a -> Column-fromUnboxedVector = UnboxedColumn Nothing--{- | O(n) Convert a list to a column. Automatically picks the best representation of a vector to store the underlying data in.--__Examples:__--@-> fromList [(1 :: Int), 2, 3, 4]-[1,2,3,4]-@--}-fromList ::- forall a.- (Columnable a, ColumnifyRep (KindOf a) a) =>- [a] -> Column-fromList = toColumnRep @(KindOf a) . VB.fromList--{- | O(n) Create a column of random elements within a range.--Takes a random number generator, a length, and a lower and upper bound for the random values.--__Examples:__--@-> import System.Random (mkStdGen)-> mkRandom (mkStdGen 42) 4 0 10-[4,2,6,5]-@--}-mkRandom ::- (RandomGen g, Columnable a, ColumnifyRep (KindOf a) a, UniformRange a) =>- g -> Int -> a -> a -> Column-mkRandom pureGen k lo hi = fromList $ go pureGen k- where- go _g 0 = []- go g n =- let- (!v, !g') = uniformR (lo, hi) g- in- v : go g' (n - 1)---- An internal helper for type errors-throwTypeMismatch ::- forall (a :: Type) (b :: Type).- (Typeable a, Typeable b) => Either DataFrameException Column-throwTypeMismatch =- Left $- TypeMismatchException- MkTypeErrorContext- { userType = Right (typeRep @b)- , expectedType = Right (typeRep @a)- , callingFunctionName = Nothing- , errorColumnName = Nothing- }---- | An internal function to map a function over the values of a column.-mapColumn ::- forall b c.- (Columnable b, Columnable c) =>- (b -> c) -> Column -> Either DataFrameException Column-mapColumn f = \case- BoxedColumn bm (col :: VB.Vector a) -> runBoxed bm col- UnboxedColumn bm (col :: VU.Vector a) -> runUnboxed bm col- c@(PackedText _ _) -> mapColumn f (materializePacked c)- c@(MergedColumn _ _) -> mapColumn f (materializeMerged c)- where- runBoxed ::- forall a.- (Columnable a) =>- Maybe Bitmap -> VB.Vector a -> Either DataFrameException Column- runBoxed bm col = case testEquality (typeRep @b) (typeRep @(Maybe a)) of- Just Refl ->- let !n = VB.length col- in Right $ case sUnbox @c of- STrue -> UnboxedColumn Nothing $- VU.generate n $ \i ->- f- ( if maybe True (`bitmapTestBit` i) bm- then Just (VB.unsafeIndex col i)- else Nothing- )- SFalse -> fromVector @c $- VB.generate n $ \i ->- f- ( if maybe True (`bitmapTestBit` i) bm- then Just (VB.unsafeIndex col i)- else Nothing- )- Nothing -> case testEquality (typeRep @a) (typeRep @b) of- Just Refl ->- Right $ case sUnbox @c of- STrue -> UnboxedColumn bm (VU.generate (VB.length col) (f . VB.unsafeIndex col))- SFalse -> case bm of- Nothing -> fromVector @c (VB.map f col)- Just _ -> BoxedColumn bm (VB.map f col)- Nothing -> throwTypeMismatch @a @b-- runUnboxed ::- forall a.- (Columnable a, VU.Unbox a) =>- Maybe Bitmap -> VU.Vector a -> Either DataFrameException Column- runUnboxed bm col = case testEquality (typeRep @b) (typeRep @(Maybe a)) of- Just Refl ->- let !n = VU.length col- in Right $ case sUnbox @c of- STrue -> UnboxedColumn Nothing $- VU.generate n $ \i ->- f- ( if maybe True (`bitmapTestBit` i) bm- then Just (VU.unsafeIndex col i)- else Nothing- )- SFalse -> fromVector @c $- VB.generate n $ \i ->- f- ( if maybe True (`bitmapTestBit` i) bm- then Just (VU.unsafeIndex col i)- else Nothing- )- Nothing -> case testEquality (typeRep @a) (typeRep @b) of- Just Refl -> Right $ case sUnbox @c of- STrue -> UnboxedColumn bm (VU.map f col)- SFalse -> case bm of- Nothing -> fromVector @c (VB.generate (VU.length col) (f . VU.unsafeIndex col))- Just _ -> BoxedColumn bm (VB.generate (VU.length col) (f . VU.unsafeIndex col))- Nothing -> throwTypeMismatch @a @b-{-# INLINEABLE mapColumn #-}---- | Applies a function that returns an unboxed result to an unboxed vector, storing the result in a column.-imapColumn ::- forall b c.- (Columnable b, Columnable c) =>- (Int -> b -> c) -> Column -> Either DataFrameException Column-imapColumn f = \case- BoxedColumn bm (col :: VB.Vector a) -> runBoxed bm col- UnboxedColumn bm (col :: VU.Vector a) -> runUnboxed bm col- c@(PackedText _ _) -> imapColumn f (materializePacked c)- c@(MergedColumn _ _) -> imapColumn f (materializeMerged c)- where- runBoxed ::- forall a.- (Columnable a) =>- Maybe Bitmap -> VB.Vector a -> Either DataFrameException Column- runBoxed bm col = case testEquality (typeRep @a) (typeRep @b) of- Just Refl -> Right $ case sUnbox @c of- STrue ->- UnboxedColumn- bm- (VU.generate (VB.length col) (\i -> f i (VB.unsafeIndex col i)))- SFalse -> BoxedColumn bm (VB.imap f col)- Nothing -> throwTypeMismatch @a @b-- runUnboxed ::- forall a.- (Columnable a, VU.Unbox a) =>- Maybe Bitmap -> VU.Vector a -> Either DataFrameException Column- runUnboxed bm col = case testEquality (typeRep @a) (typeRep @b) of- Just Refl -> Right $ case sUnbox @c of- STrue -> UnboxedColumn bm (VU.imap f col)- SFalse -> BoxedColumn bm (VB.imap f (VG.convert col))- Nothing -> throwTypeMismatch @a @b---- | O(1) Gets the number of elements in the column.-columnLength :: Column -> Int-columnLength (MergedColumn a b) = min (columnLength a) (columnLength b)-columnLength (BoxedColumn _ xs) = VB.length xs-columnLength (UnboxedColumn _ xs) = VU.length xs-columnLength (PackedText _ p) = packedLength p-{-# INLINE columnLength #-}---- | O(n) Gets the number of non-null elements in the column.-numElements :: Column -> Int-numElements (MergedColumn a b) = min (columnLength a) (columnLength b)-numElements (BoxedColumn Nothing xs) = VB.length xs-numElements (BoxedColumn (Just bm) _xs) = VU.foldl' (\acc b -> acc + popCount b) 0 bm-numElements (UnboxedColumn Nothing xs) = VU.length xs-numElements (UnboxedColumn (Just bm) _xs) = VU.foldl' (\acc b -> acc + popCount b) 0 bm-numElements (PackedText Nothing p) = packedLength p-numElements (PackedText (Just bm) _p) = VU.foldl' (\acc b -> acc + popCount b) 0 bm-{-# INLINE numElements #-}---- | O(n) Takes the first n values of a column.-takeColumn :: Int -> Column -> Column-takeColumn n (MergedColumn a b) = MergedColumn (takeColumn n a) (takeColumn n b)-takeColumn n (BoxedColumn bm xs) =- BoxedColumn (fmap (bitmapSlice 0 n) bm) (VG.take n xs)-takeColumn n (UnboxedColumn bm xs) =- UnboxedColumn (fmap (bitmapSlice 0 n) bm) (VG.take n xs)-takeColumn n (PackedText bm p) =- PackedText (fmap (bitmapSlice 0 n) bm) (packedTake n p)-{-# INLINE takeColumn #-}---- | O(n) Takes the last n values of a column.-takeLastColumn :: Int -> Column -> Column-takeLastColumn n column = sliceColumn (columnLength column - n) n column-{-# INLINE takeLastColumn #-}---- | O(n) Takes n values after a given column index.-sliceColumn :: Int -> Int -> Column -> Column-sliceColumn start n (MergedColumn a b) =- MergedColumn (sliceColumn start n a) (sliceColumn start n b)-sliceColumn start n (BoxedColumn bm xs) =- BoxedColumn (fmap (bitmapSlice start n) bm) (VG.slice start n xs)-sliceColumn start n (UnboxedColumn bm xs) =- UnboxedColumn (fmap (bitmapSlice start n) bm) (VG.slice start n xs)-sliceColumn start n c@(PackedText _ _) = sliceColumn start n (materializePacked c)-{-# INLINE sliceColumn #-}---- | O(n) Selects the elements at a given set of indices. Does not change the order.-atIndicesStable :: VU.Vector Int -> Column -> Column-atIndicesStable indexes (BoxedColumn bm column) =- BoxedColumn- ( fmap- ( \bm0 ->- buildBitmapFromValid $- VU.map (\i -> if bitmapTestBit bm0 i then 1 else 0) indexes- )- bm- )- ( VB.generate- (VU.length indexes)- ((column `VB.unsafeIndex`) . (indexes `VU.unsafeIndex`))- )-atIndicesStable indexes (UnboxedColumn bm column) =- UnboxedColumn- ( fmap- ( \bm0 ->- buildBitmapFromValid $- VU.map (\i -> if bitmapTestBit bm0 i then 1 else 0) indexes- )- bm- )- (VU.unsafeBackpermute column indexes)-atIndicesStable indexes (MergedColumn a b) =- MergedColumn (atIndicesStable indexes a) (atIndicesStable indexes b)-atIndicesStable indexes (PackedText bm p) =- PackedText- ( fmap- ( \bm0 ->- buildBitmapFromValid $- VU.map (\i -> if bitmapTestBit bm0 i then 1 else 0) indexes- )- bm- )- (packedGather indexes p)-{-# INLINE atIndicesStable #-}--{- | Like 'atIndicesStable' but treats negative indices as null.-Keeps the index vector fully unboxed (no @VB.Vector (Maybe Int)@).--}-gatherWithSentinel :: VU.Vector Int -> Column -> Column-gatherWithSentinel indices c@(MergedColumn _ _) =- gatherWithSentinel indices (materializeMerged c)-gatherWithSentinel indices col =- let !n = VU.length indices- newBm = buildBitmapFromValid $ VU.generate n $ \i ->- if VU.unsafeIndex indices i < 0 then 0 else 1- in case col of- PackedText srcBm p ->- let bm = case srcBm of- Nothing -> Just newBm- Just sb ->- Just- ( mergeBitmaps- newBm- ( buildBitmapFromValid $ VU.generate n $ \i ->- let idx = VU.unsafeIndex indices i- in if idx >= 0 && bitmapTestBit sb idx then 1 else 0- )- )- in PackedText bm (packedGather indices p)- BoxedColumn srcBm v ->- let dat = VB.generate n $ \i ->- let !idx = VU.unsafeIndex indices i- in if idx < 0 then VB.unsafeIndex v 0 else VB.unsafeIndex v idx- bm = case srcBm of- Nothing -> Just newBm- Just sb ->- Just- ( mergeBitmaps- newBm- ( buildBitmapFromValid $ VU.generate n $ \i ->- let idx = VU.unsafeIndex indices i- in if idx >= 0 && bitmapTestBit sb idx then 1 else 0- )- )- in BoxedColumn bm dat- UnboxedColumn srcBm v ->- let dat = runST $ do- mv <- VUM.new n- VG.iforM_ indices $ \i idx ->- when (idx >= 0) $ VUM.unsafeWrite mv i (VU.unsafeIndex v idx)- VU.unsafeFreeze mv- bm = case srcBm of- Nothing -> Just newBm- Just sb ->- Just- ( mergeBitmaps- newBm- ( buildBitmapFromValid $ VU.generate n $ \i ->- let idx = VU.unsafeIndex indices i- in if idx >= 0 && bitmapTestBit sb idx then 1 else 0- )- )- in UnboxedColumn bm dat-{-# INLINE gatherWithSentinel #-}---- | Internal helper to get indices in a boxed vector.-getIndices :: VU.Vector Int -> VB.Vector a -> VB.Vector a-getIndices indices xs = VB.generate (VU.length indices) (\i -> xs VB.! (indices VU.! i))-{-# INLINE getIndices #-}---- | Internal helper to get indices in an unboxed vector.-getIndicesUnboxed :: (VU.Unbox a) => VU.Vector Int -> VU.Vector a -> VU.Vector a-getIndicesUnboxed indices xs = VU.generate (VU.length indices) (\i -> xs VU.! (indices VU.! i))-{-# INLINE getIndicesUnboxed #-}--findIndices ::- forall a.- (Columnable a) =>- (a -> Bool) ->- Column ->- Either DataFrameException (VU.Vector Int)-findIndices predicate = \case- BoxedColumn _ (v :: VB.Vector b) -> run v VG.convert- UnboxedColumn _ (v :: VU.Vector b) -> run v id- c@(PackedText _ _) -> findIndices predicate (materializePacked c)- c@(MergedColumn _ _) -> findIndices predicate (materializeMerged c)- where- run ::- forall b v.- (Typeable b, VG.Vector v b, VG.Vector v Int) =>- v b ->- (v Int -> VU.Vector Int) ->- Either DataFrameException (VU.Vector Int)- run column finalize = case testEquality (typeRep @a) (typeRep @b) of- Just Refl -> Right . finalize $ VG.findIndices predicate column- Nothing ->- Left $- TypeMismatchException- MkTypeErrorContext- { userType = Right (typeRep @a)- , expectedType = Right (typeRep @b)- , callingFunctionName = Just "findIndices"- , errorColumnName = Nothing- }---- | Fold (right) column with index.-ifoldrColumn ::- forall a b.- (Columnable a, Columnable b) =>- (Int -> a -> b -> b) -> b -> Column -> Either DataFrameException b-ifoldrColumn f acc = \case- BoxedColumn _ column -> foldrWorker column- UnboxedColumn _ column -> foldrWorker column- c@(PackedText _ _) -> ifoldrColumn f acc (materializePacked c)- c@(MergedColumn _ _) -> ifoldrColumn f acc (materializeMerged c)- where- foldrWorker ::- forall c v.- (Typeable c, VG.Vector v c) =>- v c ->- Either DataFrameException b- foldrWorker vec = case testEquality (typeRep @a) (typeRep @c) of- Just Refl -> pure $ VG.ifoldr f acc vec- Nothing ->- Left $- TypeMismatchException- ( MkTypeErrorContext- { userType = Right (typeRep @a)- , expectedType = Right (typeRep @c)- , callingFunctionName = Just "ifoldrColumn"- , errorColumnName = Nothing- }- )--foldlColumn ::- forall a b.- (Columnable a, Columnable b) =>- (b -> a -> b) -> b -> Column -> Either DataFrameException b-foldlColumn f acc = \case- BoxedColumn _ column -> foldlWorker column- UnboxedColumn _ column -> foldlWorker column- c@(PackedText _ _) -> foldlColumn f acc (materializePacked c)- c@(MergedColumn _ _) -> foldlColumn f acc (materializeMerged c)- where- foldlWorker ::- forall c v.- (Typeable c, VG.Vector v c) =>- v c ->- Either DataFrameException b- foldlWorker vec = case testEquality (typeRep @a) (typeRep @c) of- Just Refl -> pure $ VG.foldl' f acc vec- Nothing ->- Left $- TypeMismatchException- ( MkTypeErrorContext- { userType = Right (typeRep @a)- , expectedType = Right (typeRep @c)- , callingFunctionName = Just "ifoldrColumn"- , errorColumnName = Nothing- }- )--foldl1Column ::- forall a.- (Columnable a) =>- (a -> a -> a) -> Column -> Either DataFrameException a-foldl1Column f = \case- BoxedColumn _ column -> foldl1Worker column- UnboxedColumn _ column -> foldl1Worker column- c@(PackedText _ _) -> foldl1Column f (materializePacked c)- c@(MergedColumn _ _) -> foldl1Column f (materializeMerged c)- where- foldl1Worker ::- forall c v.- (Typeable c, VG.Vector v c) =>- v c ->- Either DataFrameException a- foldl1Worker vec = case testEquality (typeRep @a) (typeRep @c) of- Just Refl -> pure $ VG.foldl1' f vec- Nothing ->- Left $- TypeMismatchException- ( MkTypeErrorContext- { userType = Right (typeRep @a)- , expectedType = Right (typeRep @c)- , callingFunctionName = Just "foldl1Column"- , errorColumnName = Nothing- }- )--{- | O(n) Seedless fold over groups using the first element of each group as seed.-Like 'foldDirectGroups' but for the case where no initial accumulator is available.--}-foldl1DirectGroups ::- forall a.- (Columnable a) =>- (a -> a -> a) ->- Column ->- VU.Vector Int ->- VU.Vector Int ->- Either DataFrameException Column-foldl1DirectGroups f col valueIndices offsets- | VU.length offsets <= 1 = pure $ fromVector @a VB.empty- | otherwise = case col of- UnboxedColumn _ (vec :: VU.Vector d) -> UnboxedColumn Nothing <$> foldl1Worker vec- BoxedColumn _ (vec :: VB.Vector d) -> BoxedColumn Nothing <$> foldl1Worker vec- PackedText _ _ -> foldl1DirectGroups f (materializePacked col) valueIndices offsets- MergedColumn _ _ -> foldl1DirectGroups f (materializeMerged col) valueIndices offsets- where- foldl1Worker ::- forall c v.- (Typeable c, VG.Vector v c) =>- v c ->- Either DataFrameException (v c)- foldl1Worker vec = case testEquality (typeRep @a) (typeRep @c) of- Just Refl ->- Right $- VG.generate (VU.length offsets - 1) foldGroup- where- foldGroup k =- let !s = VU.unsafeIndex offsets k- !e = VU.unsafeIndex offsets (k + 1)- !seed = VG.unsafeIndex vec (VU.unsafeIndex valueIndices s)- in go (s + 1) e seed- go !i !e !acc- | i >= e = acc- | otherwise =- go (i + 1) e $!- f acc (VG.unsafeIndex vec (VU.unsafeIndex valueIndices i))- Nothing ->- Left $- TypeMismatchException- MkTypeErrorContext- { userType = Right (typeRep @a)- , expectedType = Right (typeRep @c)- , callingFunctionName = Just "foldl1DirectGroups"- , errorColumnName = Nothing- }-{-# INLINEABLE foldl1DirectGroups #-}--{- | O(n) fold over groups by scanning the column linearly (rowToGroup[i] = group-of row i). Random writes hit the small per-group accumulator array; when @acc@ is-unboxable that array is unboxed, avoiding pointer indirection.--}-foldLinearGroups ::- forall b acc.- (Columnable b, Columnable acc) =>- (acc -> b -> acc) ->- acc ->- Column ->- VU.Vector Int ->- Int ->- Either DataFrameException Column-foldLinearGroups f seed col rowToGroup nGroups- | nGroups == 0 = Right (fromVector @acc VB.empty)- | otherwise = case col of- UnboxedColumn _ (vec :: VU.Vector d) -> foldLinearWorker vec- BoxedColumn _ (vec :: VB.Vector d) -> foldLinearWorker vec- PackedText _ _ ->- foldLinearGroups f seed (materializePacked col) rowToGroup nGroups- MergedColumn _ _ ->- foldLinearGroups f seed (materializeMerged col) rowToGroup nGroups- where- foldLinearWorker ::- forall c v.- (Typeable c, VG.Vector v c) =>- v c ->- Either DataFrameException Column- foldLinearWorker vec = case testEquality (typeRep @b) (typeRep @c) of- Just Refl ->- Right $- unsafePerformIO $- runWith- ( \readAt writeAt ->- VG.iforM_ vec $ \row x -> do- let !k = VG.unsafeIndex rowToGroup row- cur <- readAt k- writeAt k $! f cur x- )- Nothing ->- Left $- TypeMismatchException- MkTypeErrorContext- { userType = Right (typeRep @b)- , expectedType = Right (typeRep @c)- , callingFunctionName = Just "foldLinearGroups"- , errorColumnName = Nothing- }-- runWith :: ((Int -> IO acc) -> (Int -> acc -> IO ()) -> IO ()) -> IO Column- runWith body = case sUnbox @acc of- STrue -> do- accs <- VUM.replicate nGroups seed- body (VUM.unsafeRead accs) (VUM.unsafeWrite accs)- UnboxedColumn Nothing <$> VU.unsafeFreeze accs- SFalse -> do- accs <- VBM.replicate nGroups seed- body (VBM.unsafeRead accs) (VBM.unsafeWrite accs)- fromVector @acc <$> VB.unsafeFreeze accs- {-# INLINE runWith #-}-{-# INLINEABLE foldLinearGroups #-}--headColumn :: forall a. (Columnable a) => Column -> Either DataFrameException a-headColumn = \case- BoxedColumn _ col -> headWorker col- UnboxedColumn _ col -> headWorker col- c@(PackedText _ _) -> headColumn (materializePacked c)- c@(MergedColumn _ _) -> headColumn (mergedHead c)- where- headWorker ::- forall c v.- (Typeable c, VG.Vector v c) =>- v c ->- Either DataFrameException a- headWorker vec = case testEquality (typeRep @a) (typeRep @c) of- Just Refl ->- if VG.null vec- then Left (EmptyDataSetException "headColumn")- else pure (VG.head vec)- Nothing ->- Left $- TypeMismatchException- ( MkTypeErrorContext- { userType = Right (typeRep @a)- , expectedType = Right (typeRep @c)- , callingFunctionName = Just "headColumn"- , errorColumnName = Nothing- }- )---- | An internal, column version of zip.-zipColumns :: Column -> Column -> Column-zipColumns l@(MergedColumn _ _) r = zipColumns (materializeMerged l) r-zipColumns l r@(MergedColumn _ _) = zipColumns l (materializeMerged r)-zipColumns l@(PackedText _ _) r = zipColumns (materializePacked l) r-zipColumns l r@(PackedText _ _) = zipColumns l (materializePacked r)-zipColumns (BoxedColumn _ column) (BoxedColumn _ other) = BoxedColumn Nothing (VG.zip column other)-zipColumns (BoxedColumn _ column) (UnboxedColumn _ other) =- BoxedColumn- Nothing- ( VB.generate- (min (VG.length column) (VG.length other))- (\i -> (column VG.! i, other VG.! i))- )-zipColumns (UnboxedColumn _ column) (BoxedColumn _ other) =- BoxedColumn- Nothing- ( VB.generate- (min (VG.length column) (VG.length other))- (\i -> (column VG.! i, other VG.! i))- )-zipColumns (UnboxedColumn _ column) (UnboxedColumn _ other) = UnboxedColumn Nothing (VG.zip column other)-{-# INLINE zipColumns #-}--{- | Merge two columns using `These`. O(1): the sides are kept in their-native representation and 'These' values materialize on element access.--}-mergeColumns :: Column -> Column -> Column-mergeColumns = MergedColumn-{-# INLINE mergeColumns #-}---- | Decode a 'MergedColumn' into the eager @BoxedColumn (These a b)@ form.-materializeMerged :: Column -> Column-materializeMerged (MergedColumn colA colB) =- mergeEager (materializeMerged colA) (materializeMerged colB)-materializeMerged c = c--mergedHead :: Column -> Column-mergedHead (MergedColumn a b) =- materializeMerged (MergedColumn (takeColumn 1 a) (takeColumn 1 b))-mergedHead c = c--{- | The eager element-wise merge ('These' per row, boxed). Bitmaps are-honored for every representation pair: a null side yields 'This'/'That',-both-null is an error (the join kernels never produce such a row).--}-mergeEager :: Column -> Column -> Column-mergeEager colA colB = case (colA, colB) of- (MergedColumn a b, _) -> mergeEager (mergeEager a b) colB- (_, MergedColumn a b) -> mergeEager colA (mergeEager a b)- (PackedText _ _, _) -> mergeEager (materializePacked colA) colB- (_, PackedText _ _) -> mergeEager colA (materializePacked colB)- (BoxedColumn bmA c1, BoxedColumn bmB c2) ->- merged bmA bmB (VG.length c1) (VG.length c2) (c1 VG.!) (c2 VG.!)- (BoxedColumn bmA c1, UnboxedColumn bmB c2) ->- merged bmA bmB (VG.length c1) (VG.length c2) (c1 VG.!) (c2 VG.!)- (UnboxedColumn bmA c1, BoxedColumn bmB c2) ->- merged bmA bmB (VG.length c1) (VG.length c2) (c1 VG.!) (c2 VG.!)- (UnboxedColumn bmA c1, UnboxedColumn bmB c2) ->- merged bmA bmB (VG.length c1) (VG.length c2) (c1 VG.!) (c2 VG.!)- where- merged ::- (Columnable a, Columnable b) =>- Maybe Bitmap ->- Maybe Bitmap ->- Int ->- Int ->- (Int -> a) ->- (Int -> b) ->- Column- merged bmA bmB lenA lenB atA atB =- BoxedColumn Nothing $ VB.generate (min lenA lenB) $ \i ->- case (validAt bmA i, validAt bmB i) of- (True, True) -> These (atA i) (atB i)- (True, False) -> This (atA i)- (False, True) -> That (atB i)- (False, False) -> error "mergeColumns: both null"- validAt mbm i = maybe True (`bitmapTestBit` i) mbm- {-# INLINE validAt #-}---- | An internal, column version of zipWith.-zipWithColumns ::- forall a b c.- (Columnable a, Columnable b, Columnable c) =>- (a -> b -> c) -> Column -> Column -> Either DataFrameException Column-zipWithColumns f (UnboxedColumn bmL (column :: VU.Vector d)) (UnboxedColumn bmR (other :: VU.Vector e)) = case testEquality (typeRep @a) (typeRep @d) of- Just Refl -> case testEquality (typeRep @b) (typeRep @e) of- Just Refl- | isNothing bmL- , isNothing bmR ->- pure $ case sUnbox @c of- STrue -> UnboxedColumn Nothing (VU.zipWith f column other)- SFalse -> fromVector $ VB.zipWith f (VG.convert column) (VG.convert other)- _ -> zipWithColumnsGeneral f (UnboxedColumn bmL column) (UnboxedColumn bmR other)- Nothing -> zipWithColumnsGeneral f (UnboxedColumn bmL column) (UnboxedColumn bmR other)--- TODO: mchavinda - reuse pattern from interpret where we augment the--- error at the end.-zipWithColumns f left right = zipWithColumnsGeneral f left right--zipWithColumnsGeneral ::- forall a b c.- (Columnable a, Columnable b, Columnable c) =>- (a -> b -> c) -> Column -> Column -> Either DataFrameException Column-zipWithColumnsGeneral f left right = case toVector @a left of- Left (TypeMismatchException context) ->- Left $- TypeMismatchException (context{callingFunctionName = Just "zipWithColumns"})- Left e -> Left e- Right left' -> case toVector @b right of- Left (TypeMismatchException context) ->- Left $- TypeMismatchException (context{callingFunctionName = Just "zipWithColumns"})- Left e -> Left e- Right right' -> pure $ fromVector $ VB.zipWith f left' right'-{-# INLINE zipWithColumnsGeneral #-}-{-# INLINE zipWithColumns #-}---- writeColumn and freezeColumn' (CSV-ingest helpers) moved to--- DataFrame.IO.Internal.MutableColumn so the core column module does not--- need to depend on DataFrame.Internal.Parsing.--{- | Freeze a mutable column into an @Either Text a@ column: every recorded-null position becomes @Left rawText@ (preserving the original input), every-other position becomes @Right v@. Used by CSV readers under 'EitherRead' mode.--}-freezeColumnEither :: [(Int, T.Text)] -> MutableColumn -> IO Column-freezeColumnEither nulls (MBoxedColumn col) = do- frozen <- VB.unsafeFreeze col- let nullMap = nulls- pure $- BoxedColumn Nothing $- VB.imap- ( \i v -> case lookup i nullMap of- Just t -> Left t- Nothing -> Right v- )- frozen-freezeColumnEither nulls (MUnboxedColumn col) = do- c <- VU.unsafeFreeze col- let nullMap = nulls- pure $- BoxedColumn Nothing $- VB.generate (VU.length c) $ \i ->- case lookup i nullMap of- Just t -> Left t- Nothing -> Right (c VU.! i)-{-# INLINE freezeColumnEither #-}--{- | Promote a non-nullable column to a nullable one (add an all-valid bitmap).-No-op when already nullable.--}-ensureOptional :: Column -> Column-ensureOptional c@(MergedColumn _ _) = ensureOptional (materializeMerged c)-ensureOptional c@(BoxedColumn (Just _) _) = c-ensureOptional (BoxedColumn Nothing col) =- BoxedColumn (Just (allValidBitmap (VB.length col))) col-ensureOptional c@(UnboxedColumn (Just _) _) = c-ensureOptional (UnboxedColumn Nothing col) =- UnboxedColumn (Just (allValidBitmap (VU.length col))) col-ensureOptional c@(PackedText (Just _) _) = c-ensureOptional (PackedText Nothing p) =- PackedText (Just (allValidBitmap (packedLength p))) p---- | Fills the end of a column, up to n, with null rows. Does nothing if column has length >= n.-expandColumn :: Int -> Column -> Column-expandColumn n c@(MergedColumn a b)- | n <= min (columnLength a) (columnLength b) = c- | otherwise = expandColumn n (materializeMerged c)-expandColumn n c@(PackedText _ p)- | n <= packedLength p = c- | otherwise = expandColumn n (materializePacked c)-expandColumn n column@(BoxedColumn bm col)- | n <= VG.length col = column- | otherwise =- let extra = n - VG.length col- newBm = case bm of- Nothing -> Just (buildBitmapFromNulls n [VG.length col .. n - 1])- Just b ->- Just- (bitmapConcat (VG.length col) b extra (VU.replicate ((extra + 7) `shiftR` 3) 0))- newCol = col <> VB.replicate extra (errorWithoutStackTrace "expandColumn: null slot")- in BoxedColumn newBm newCol-expandColumn n column@(UnboxedColumn bm col)- | n <= VG.length col = column- | otherwise =- let extra = n - VG.length col- newBm = case bm of- Nothing -> Just (buildBitmapFromNulls n [VG.length col .. n - 1])- Just b ->- Just- (bitmapConcat (VG.length col) b extra (VU.replicate ((extra + 7) `shiftR` 3) 0))- newCol = runST $ do- mv <- VUM.new n- VU.imapM_ (VUM.unsafeWrite mv) col- VU.unsafeFreeze mv- in UnboxedColumn newBm newCol---- | Fills the beginning of a column, up to n, with null rows. Does nothing if column has length >= n.-leftExpandColumn :: Int -> Column -> Column-leftExpandColumn n c@(MergedColumn a b)- | n <= min (columnLength a) (columnLength b) = c- | otherwise = leftExpandColumn n (materializeMerged c)-leftExpandColumn n c@(PackedText _ p)- | n <= packedLength p = c- | otherwise = leftExpandColumn n (materializePacked c)-leftExpandColumn n column@(BoxedColumn bm col)- | n <= VG.length col = column- | otherwise =- let extra = n - VG.length col- origLen = VG.length col- newBm = case bm of- Nothing -> Just (buildBitmapFromNulls n [0 .. extra - 1])- Just b ->- let nullPart = VU.replicate ((extra + 7) `shiftR` 3) 0- in Just (bitmapConcat extra nullPart origLen b)- newCol =- VB.replicate extra (errorWithoutStackTrace "leftExpandColumn: null slot") <> col- in BoxedColumn newBm newCol-leftExpandColumn n column@(UnboxedColumn bm col)- | n <= VG.length col = column- | otherwise =- let extra = n - VG.length col- origLen = VG.length col- newBm = case bm of- Nothing -> Just (buildBitmapFromNulls n [0 .. extra - 1])- Just b ->- let nullPart = VU.replicate ((extra + 7) `shiftR` 3) 0- in Just (bitmapConcat extra nullPart origLen b)- newCol = runST $ do- mv <- VUM.new n- VU.imapM_ (\i x -> VUM.unsafeWrite mv (extra + i) x) col- VU.unsafeFreeze mv- in UnboxedColumn newBm newCol--{- | Concatenates two columns.-Returns Nothing if the columns are of different types.--}-concatColumns :: Column -> Column -> Either DataFrameException Column-concatColumns left right = case (left, right) of- (MergedColumn _ _, _) -> concatColumns (materializeMerged left) right- (_, MergedColumn _ _) -> concatColumns left (materializeMerged right)- (PackedText _ _, _) -> concatColumns (materializePacked left) right- (_, PackedText _ _) -> concatColumns left (materializePacked right)- (BoxedColumn bmL l, BoxedColumn bmR r) -> case testEquality (typeOf l) (typeOf r) of- Just Refl ->- let newBm = case (bmL, bmR) of- (Nothing, Nothing) -> Nothing- (Just bl, Nothing) ->- Just- (bitmapConcat (VB.length l) bl (VB.length r) (allValidBitmap (VB.length r)))- (Nothing, Just br) ->- Just- (bitmapConcat (VB.length l) (allValidBitmap (VB.length l)) (VB.length r) br)- (Just bl, Just br) -> Just (bitmapConcat (VB.length l) bl (VB.length r) br)- in pure (BoxedColumn newBm (l <> r))- Nothing -> Left (mismatchErr (typeOf r) (typeOf l))- (UnboxedColumn bmL l, UnboxedColumn bmR r) -> case testEquality (typeOf l) (typeOf r) of- Just Refl ->- let newBm = case (bmL, bmR) of- (Nothing, Nothing) -> Nothing- (Just bl, Nothing) ->- Just- (bitmapConcat (VU.length l) bl (VU.length r) (allValidBitmap (VU.length r)))- (Nothing, Just br) ->- Just- (bitmapConcat (VU.length l) (allValidBitmap (VU.length l)) (VU.length r) br)- (Just bl, Just br) -> Just (bitmapConcat (VU.length l) bl (VU.length r) br)- in pure (UnboxedColumn newBm (l <> r))- Nothing -> Left (mismatchErr (typeOf r) (typeOf l))- _ -> Left (mismatchErr (typeOf right) (typeOf left))- where- mismatchErr ::- forall (x :: Type) (y :: Type). TypeRep x -> TypeRep y -> DataFrameException- mismatchErr ta tb =- withTypeable ta $- withTypeable tb $- TypeMismatchException- ( MkTypeErrorContext- { userType = Right ta- , expectedType = Right tb- , callingFunctionName = Just "concatColumns"- , errorColumnName = Nothing- }- )--{- | Like 'concatColumns' but also combines columns of different types by wrapping-values in 'Either' (e.g. @[1,2]@ and @["a","b"]@ become-@[Left 1, Left 2, Right "a", Right "b"]@).--}--{- | O(n) Concatenate a list of same-type columns in a single allocation.-All columns must have the same constructor and element type (as they will-within a single Parquet column). Calls 'error' on mismatch.--}-concatManyColumns :: [Column] -> Column-concatManyColumns [] = fromList ([] :: [Maybe Int])-concatManyColumns [c] = c-concatManyColumns all'- | any isMergedColumn all' =- concatManyColumns (map materializeMerged all')- | any isPackedText all' =- concatManyColumns (map materializePacked all')-concatManyColumns (c0 : cs) = case c0 of- BoxedColumn bm0 v0 ->- let getCol (BoxedColumn bm v) = case testEquality (typeOf v0) (typeOf v) of- Just Refl -> (bm, v)- Nothing -> error "concatManyColumns: BoxedColumn type mismatch"- getCol _ = error "concatManyColumns: column constructor mismatch"- rest = map getCol cs- allVecs = v0 : map snd rest- allBms = bm0 : map fst rest- newBm- | all isNothing allBms = Nothing- | otherwise =- let pairs = zip allVecs allBms- expandedBms = map (\(v, mb) -> fromMaybe (allValidBitmap (VB.length v)) mb) pairs- go b1 n1 b2 n2 = bitmapConcat n1 b1 n2 b2- concatBms [] = VU.empty- concatBms [(b, _v)] = b- concatBms ((b1, v1) : (b2, v2) : rest') =- let merged = go b1 (VB.length v1) b2 (VB.length v2)- in concatBms ((merged, v1 <> v2) : rest')- in Just $ concatBms (zip expandedBms allVecs)- in BoxedColumn newBm (VB.concat allVecs)- UnboxedColumn bm0 v0 ->- let getCol (UnboxedColumn bm v) = case testEquality (typeOf v0) (typeOf v) of- Just Refl -> (bm, v)- Nothing -> error "concatManyColumns: UnboxedColumn type mismatch"- getCol _ = error "concatManyColumns: column constructor mismatch"- rest = map getCol cs- allVecs = v0 : map snd rest- allBms = bm0 : map fst rest- newBm- | all isNothing allBms = Nothing- | otherwise =- let pairs = zip allVecs allBms- expandedBms = map (\(v, mb) -> fromMaybe (allValidBitmap (VU.length v)) mb) pairs- go b1 n1 b2 n2 = bitmapConcat n1 b1 n2 b2- concatBms [] = VU.empty- concatBms [(b, _)] = b- concatBms ((b1, v1) : (b2, v2) : rest') =- let merged = go b1 (VU.length v1) b2 (VU.length v2)- in concatBms ((merged, v1 <> v2) : rest')- in Just $ concatBms (zip expandedBms allVecs)- in UnboxedColumn newBm (VU.concat allVecs)- PackedText _ _ -> concatManyColumns (map materializePacked (c0 : cs))- MergedColumn _ _ -> concatManyColumns (map materializeMerged (c0 : cs))--concatColumnsEither :: Column -> Column -> Column-concatColumnsEither l@(MergedColumn _ _) r =- concatColumnsEither (materializeMerged l) r-concatColumnsEither l r@(MergedColumn _ _) =- concatColumnsEither l (materializeMerged r)-concatColumnsEither l@(PackedText _ _) r = concatColumnsEither (materializePacked l) r-concatColumnsEither l r@(PackedText _ _) = concatColumnsEither l (materializePacked r)-concatColumnsEither (BoxedColumn bmL left) (BoxedColumn bmR right) = case testEquality (typeOf left) (typeOf right) of- Nothing ->- BoxedColumn Nothing $ fmap Left left <> fmap Right right- Just Refl ->- let newBm = case (bmL, bmR) of- (Nothing, Nothing) -> Nothing- (Just bl, Nothing) ->- Just- ( bitmapConcat- (VB.length left)- bl- (VB.length right)- (allValidBitmap (VB.length right))- )- (Nothing, Just br) ->- Just- ( bitmapConcat- (VB.length left)- (allValidBitmap (VB.length left))- (VB.length right)- br- )- (Just bl, Just br) -> Just (bitmapConcat (VB.length left) bl (VB.length right) br)- in BoxedColumn newBm $ left <> right-concatColumnsEither (UnboxedColumn bmL left) (UnboxedColumn bmR right) = case testEquality (typeOf left) (typeOf right) of- Nothing ->- BoxedColumn Nothing $- fmap Left (VG.convert left) <> fmap Right (VG.convert right)- Just Refl ->- let newBm = case (bmL, bmR) of- (Nothing, Nothing) -> Nothing- (Just bl, Nothing) ->- Just- ( bitmapConcat- (VU.length left)- bl- (VU.length right)- (allValidBitmap (VU.length right))- )- (Nothing, Just br) ->- Just- ( bitmapConcat- (VU.length left)- (allValidBitmap (VU.length left))- (VU.length right)- br- )- (Just bl, Just br) -> Just (bitmapConcat (VU.length left) bl (VU.length right) br)- in UnboxedColumn newBm $ left <> right-concatColumnsEither (BoxedColumn _ left) (UnboxedColumn _ right) =- BoxedColumn Nothing $ fmap Left left <> fmap Right (VG.convert right)-concatColumnsEither (UnboxedColumn _ left) (BoxedColumn _ right) =- BoxedColumn Nothing $ fmap Left (VG.convert left) <> fmap Right right---- | Allocate a mutable column of size @n@ matching the constructor/type of the given column.-newMutableColumn :: Int -> Column -> IO MutableColumn-newMutableColumn n (BoxedColumn _ (_ :: VB.Vector a)) =- MBoxedColumn <$> (VBM.new n :: IO (VBM.IOVector a))-newMutableColumn n (UnboxedColumn _ (_ :: VU.Vector a)) =- MUnboxedColumn <$> (VUM.new n :: IO (VUM.IOVector a))-newMutableColumn n c@(PackedText _ _) = newMutableColumn n (materializePacked c)-newMutableColumn n c@(MergedColumn _ _) = newMutableColumn n (materializeMerged c)---- | Copy a column chunk into a mutable column starting at offset @off@.-copyIntoMutableColumn :: MutableColumn -> Int -> Column -> IO ()-copyIntoMutableColumn mv off c@(MergedColumn _ _) =- copyIntoMutableColumn mv off (materializeMerged c)-copyIntoMutableColumn (MBoxedColumn (mv :: VBM.IOVector b)) off (BoxedColumn _ (v :: VB.Vector a)) =- case testEquality (typeRep @a) (typeRep @b) of- Just Refl -> VG.imapM_ (\i x -> VBM.unsafeWrite mv (off + i) x) v- Nothing -> error "copyIntoMutableColumn: Boxed type mismatch"-copyIntoMutableColumn (MUnboxedColumn (mv :: VUM.IOVector b)) off (UnboxedColumn _ (v :: VU.Vector a)) =- case testEquality (typeRep @a) (typeRep @b) of- Just Refl -> VG.imapM_ (\i x -> VUM.unsafeWrite mv (off + i) x) v- Nothing -> error "copyIntoMutableColumn: Unboxed type mismatch"-copyIntoMutableColumn mc off c@(PackedText _ _) =- copyIntoMutableColumn mc off (materializePacked c)-copyIntoMutableColumn _ _ _ =- error "copyIntoMutableColumn: constructor mismatch"---- | Freeze a mutable column into an immutable column.-freezeMutableColumn :: MutableColumn -> IO Column-freezeMutableColumn (MBoxedColumn mv) = BoxedColumn Nothing <$> VB.unsafeFreeze mv-freezeMutableColumn (MUnboxedColumn mv) = UnboxedColumn Nothing <$> VU.unsafeFreeze mv--{- | O(n) Converts a column to a list. Throws an exception if the wrong type is specified.--__Examples:__--@-> column = fromList [(1 :: Int), 2, 3, 4]-> toList @Int column-[1,2,3,4]-> toList @Double column-exception: ...-@--}-toList :: forall a. (Columnable a) => Column -> [a]-toList xs = case toVector @a xs of- Left err -> throw err- Right val -> VB.toList val--{- | Type-safe conversion of a column to a vector of element type @a@ (specify via-type application); 'Left' 'TypeMismatchException' when the column's type differs.-->>> toVector @Int @VU.Vector column-Right (unboxed vector of Ints)-->>> toVector @Text @VB.Vector column-Right (boxed vector of Text)--}-toVector ::- forall a v.- (VG.Vector v a, Columnable a) => Column -> Either DataFrameException (v a)-toVector col = case col of- PackedText _ _ -> toVector (materializePacked col)- MergedColumn _ _ -> toVector (materializeMerged col)- BoxedColumn bm (inner :: VB.Vector c) ->- -- Check if user wants Maybe c (nullable) or c directly- case testEquality (typeRep @a) (typeRep @c) of- Just Refl -> Right $ VG.convert inner- Nothing ->- -- Try: a = Maybe c- case testEquality (typeRep @a) (typeRep @(Maybe c)) of- Just Refl ->- -- Use VB.generate to avoid fusion forcing null slots- let !n = VB.length inner- maybeVec = case bm of- Nothing -> VB.generate n (Just . VB.unsafeIndex inner)- Just bitmap -> VB.generate n $ \i ->- if bitmapTestBit bitmap i then Just (VB.unsafeIndex inner i) else Nothing- in Right $ VG.convert maybeVec- Nothing ->- Left $- TypeMismatchException- ( MkTypeErrorContext- { userType = Right (typeRep @a)- , expectedType = Right (typeRep @c)- , callingFunctionName = Just "toVector"- , errorColumnName = Nothing- }- )- UnboxedColumn bm (inner :: VU.Vector c) ->- case testEquality (typeRep @a) (typeRep @c) of- Just Refl -> Right $ VG.convert inner- Nothing ->- case testEquality (typeRep @a) (typeRep @(Maybe c)) of- Just Refl ->- let maybeVec = case bm of- Nothing -> VB.generate (VU.length inner) (Just . VU.unsafeIndex inner)- Just bitmap -> VB.generate (VU.length inner) $ \i ->- if bitmapTestBit bitmap i then Just (VU.unsafeIndex inner i) else Nothing- in Right $ VG.convert maybeVec- Nothing ->- Left $- TypeMismatchException- ( MkTypeErrorContext- { userType = Right (typeRep @a)- , expectedType = Right (typeRep @c)- , callingFunctionName = Just "toVector"- , errorColumnName = Nothing- }- )---- Some common types we will use for numerical computing.--{- | Convert a column to an unboxed 'Double' vector, coercing numeric types-('realToFrac' for floats, 'fromIntegral' for integrals; nulls become @NaN@).-'Left' 'TypeMismatchException' when the column is not numeric.--}-toDoubleVector :: Column -> Either DataFrameException (VU.Vector Double)-toDoubleVector column =- case column of- PackedText _ _ -> toDoubleVector (materializePacked column)- MergedColumn _ _ -> toDoubleVector (materializeMerged column)- UnboxedColumn bm (f :: VU.Vector a) -> case testEquality (typeRep @a) (typeRep @Double) of- Just Refl -> case bm of- Nothing -> Right f- Just bitmap -> Right $ VU.imap (\i x -> if bitmapTestBit bitmap i then x else read "NaN") f- Nothing -> case sFloating @a of- STrue ->- Right- ( VU.imap- ( \i x -> case bm of- Just bitmap | not (bitmapTestBit bitmap i) -> read "NaN"- _ -> realToFrac x- )- f- )- SFalse -> case sIntegral @a of- STrue ->- Right- ( VU.imap- ( \i x -> case bm of- Just bitmap | not (bitmapTestBit bitmap i) -> read "NaN"- _ -> fromIntegral x- )- f- )- SFalse ->- Left $- TypeMismatchException- ( MkTypeErrorContext- { userType = Right (typeRep @Double)- , expectedType = Right (typeRep @a)- , callingFunctionName = Just "toDoubleVector"- , errorColumnName = Nothing- }- )- BoxedColumn bm (f :: VB.Vector a) -> case testEquality (typeRep @a) (typeRep @Integer) of- Just Refl ->- Right- ( VB.convert $- VB.imap- ( \i x -> case bm of- Just bitmap | not (bitmapTestBit bitmap i) -> read "NaN"- _ -> fromIntegral x- )- f- )- Nothing ->- Left $- TypeMismatchException- ( MkTypeErrorContext- { userType = Right (typeRep @Double)- , expectedType = Left (columnTypeString column) :: Either String (TypeRep ())- , callingFunctionName = Just "toDoubleVector"- , errorColumnName = Nothing- }- )--{- | Convert a column to an unboxed 'Float' vector, coercing numeric types (nulls-become @NaN@); 'Left' 'TypeMismatchException' when not numeric. Converting from-'Double' may lose precision.--}-toFloatVector :: Column -> Either DataFrameException (VU.Vector Float)-toFloatVector column =- case column of- PackedText _ _ -> toFloatVector (materializePacked column)- MergedColumn _ _ -> toFloatVector (materializeMerged column)- UnboxedColumn bm (f :: VU.Vector a) -> case testEquality (typeRep @a) (typeRep @Float) of- Just Refl -> case bm of- Nothing -> Right f- Just bitmap -> Right $ VU.imap (\i x -> if bitmapTestBit bitmap i then x else read "NaN") f- Nothing -> case sFloating @a of- STrue ->- Right- ( VU.imap- ( \i x -> case bm of- Just bitmap | not (bitmapTestBit bitmap i) -> read "NaN"- _ -> realToFrac x- )- f- )- SFalse -> case sIntegral @a of- STrue ->- Right- ( VU.imap- ( \i x -> case bm of- Just bitmap | not (bitmapTestBit bitmap i) -> read "NaN"- _ -> fromIntegral x- )- f- )- SFalse ->- Left $- TypeMismatchException- ( MkTypeErrorContext- { userType = Right (typeRep @Float)- , expectedType = Right (typeRep @a)- , callingFunctionName = Just "toFloatVector"- , errorColumnName = Nothing- }- )- BoxedColumn bm (f :: VB.Vector a) -> case testEquality (typeRep @a) (typeRep @Integer) of- Just Refl ->- Right- ( VB.convert $- VB.imap- ( \i x -> case bm of- Just bitmap | not (bitmapTestBit bitmap i) -> read "NaN"- _ -> fromIntegral x- )- f- )- Nothing ->- Left $- TypeMismatchException- ( MkTypeErrorContext- { userType = Right (typeRep @Float)- , expectedType = Left (columnTypeString column) :: Either String (TypeRep ())- , callingFunctionName = Just "toFloatVector"- , errorColumnName = Nothing- }- )--{- | Convert a column to an unboxed 'Int' vector, coercing numeric types-(floats are 'round'ed via banker's rounding); 'Left' 'TypeMismatchException'-when the column is not numeric. Does not support nullable columns.--}-toIntVector :: Column -> Either DataFrameException (VU.Vector Int)-toIntVector column =- case column of- PackedText _ _ -> toIntVector (materializePacked column)- MergedColumn _ _ -> toIntVector (materializeMerged column)- UnboxedColumn _ (f :: VU.Vector a) -> case testEquality (typeRep @a) (typeRep @Int) of- Just Refl -> Right f- Nothing -> case sFloating @a of- STrue -> Right (VU.map (round . (realToFrac :: a -> Double)) f)- SFalse -> case sIntegral @a of- STrue -> Right (VU.map fromIntegral f)- SFalse ->- Left $- TypeMismatchException- ( MkTypeErrorContext- { userType = Right (typeRep @Int)- , expectedType = Right (typeRep @a)- , callingFunctionName = Just "toIntVector"- , errorColumnName = Nothing- }- )- BoxedColumn _ (f :: VB.Vector a) -> case testEquality (typeRep @a) (typeRep @Integer) of- Just Refl -> Right (VB.convert $ VB.map fromIntegral f)- Nothing ->- Left $- TypeMismatchException- ( MkTypeErrorContext- { userType = Right (typeRep @Int)- , expectedType = Left (columnTypeString column) :: Either String (TypeRep ())- , callingFunctionName = Just "toIntVector"- , errorColumnName = Nothing- }- )--toUnboxedVector ::- forall a.- (Columnable a, VU.Unbox a) => Column -> Either DataFrameException (VU.Vector a)-toUnboxedVector column =- case column of- UnboxedColumn _ (f :: VU.Vector b) -> case testEquality (typeRep @a) (typeRep @b) of- Just Refl -> Right f- Nothing ->- Left $- TypeMismatchException- ( MkTypeErrorContext- { userType = Right (typeRep @a)- , expectedType = Right (typeRep @b)- , callingFunctionName = Just "toUnboxedVector"- , errorColumnName = Nothing- }- )- _ ->- Left $- TypeMismatchException- ( MkTypeErrorContext- { userType = Right (typeRep @a)- , expectedType = Left (columnTypeString column) :: Either String (TypeRep ())- , callingFunctionName = Just "toUnboxedVector"- , errorColumnName = Nothing- }- )-{-# INLINE toUnboxedVector #-}---- Shared finaliser for the two parseUnboxedColumn* helpers. Freezes--- the mutable data vector, and only materialises the bitmap when the--- column actually had nulls.-{-# INLINE finalizeParseResult #-}-finalizeParseResult ::- (VU.Unbox a) =>- VUM.STVector s a ->- VUM.STVector s Word8 ->- Bool ->- ST s (Maybe (Maybe Bitmap, VU.Vector a))-finalizeParseResult values vmask anyNull- | anyNull = do- vs <- VU.unsafeFreeze values- vm <- VU.unsafeFreeze vmask- return (Just (Just (buildBitmapFromValid vm), vs))- | otherwise = do- vs <- VU.unsafeFreeze values- return (Just (Nothing, vs))+{- |+Umbrella re-export of the column implementation. The module is split by+concern:++* "DataFrame.Internal.Column.Types" — type-level machinery ('Rep', 'SBool', ...)+* "DataFrame.Internal.Column.Base" — the 'Column' GADT and core definitions+* "DataFrame.Internal.Column.Properties" — predicates and introspection+* "DataFrame.Internal.Column.Conversion" — vector\/list conversions+* "DataFrame.Internal.Column.Operations" — bulk transformations++Import this module to get the whole surface; import a submodule directly when+you only need one layer.+-}+module DataFrame.Internal.Column (+ module DataFrame.Internal.Column.Base,+ module DataFrame.Internal.Column.Conversion,+ module DataFrame.Internal.Column.Operations,+ module DataFrame.Internal.Column.Properties,+ module DataFrame.Internal.Column.Types,+) where++import DataFrame.Internal.Column.Base+import DataFrame.Internal.Column.Conversion+import DataFrame.Internal.Column.Operations+import DataFrame.Internal.Column.Properties+import DataFrame.Internal.Column.Types
+ src-internal/DataFrame/Internal/Column/Base.hs view
@@ -0,0 +1,333 @@+{-# LANGUAGE AllowAmbiguousTypes #-}+{-# LANGUAGE BangPatterns #-}+{-# LANGUAGE ConstraintKinds #-}+{-# LANGUAGE DataKinds #-}+{-# LANGUAGE FlexibleContexts #-}+{-# LANGUAGE FlexibleInstances #-}+{-# LANGUAGE GADTs #-}+{-# LANGUAGE InstanceSigs #-}+{-# LANGUAGE MultiParamTypeClasses #-}+{-# LANGUAGE OverloadedStrings #-}+{-# LANGUAGE PolyKinds #-}+{-# LANGUAGE RankNTypes #-}+{-# LANGUAGE ScopedTypeVariables #-}+{-# LANGUAGE TypeApplications #-}+{-# LANGUAGE TypeFamilies #-}+{-# LANGUAGE UndecidableInstances #-}++{- |+Core column definitions: the type-erased 'Column' GADT and its mutable/typed+companions, the 'Columnable' constraint, and the representation-level+primitives ('materializePacked', 'materializeMerged') that the non-orphan+'Show'/'Eq' instances depend on.++Predicates live in "DataFrame.Internal.Column.Properties", vector/list+conversions in "DataFrame.Internal.Column.Conversion", and bulk transformations+in "DataFrame.Internal.Column.Operations".+-}+module DataFrame.Internal.Column.Base where++import qualified Data.Vector as VB+import qualified Data.Vector.Generic as VG+import qualified Data.Vector.Mutable as VBM+import qualified Data.Vector.Unboxed as VU+import qualified Data.Vector.Unboxed.Mutable as VUM++import Control.Monad (forM_)+import Control.Monad.ST (runST)+import Data.Maybe (fromMaybe, isNothing)+import Data.Type.Equality (TestEquality (..))+import DataFrame.Internal.Column.Bitmap+import DataFrame.Internal.Column.Types+import DataFrame.Internal.Data.PackedText (+ PackedTextData (..),+ packedIndexText,+ packedLength,+ packedSlice,+ packedTake,+ sliceEqBytes,+ )+import Type.Reflection (typeRep, type (:~:) (Refl))++-- | Constraint synonym for what we can put into columns.+type Columnable a =+ ( Columnable' a+ , ColumnifyRep (KindOf a) a+ , UnboxIf a+ , IntegralIf a+ , FloatingIf a+ , SBoolI (Unboxable a)+ , SBoolI (Numeric a)+ , SBoolI (IntegralTypes a)+ , SBoolI (FloatingTypes a)+ )++{- | Type-erased column GADT. Pattern-matching on the constructor recovers the+representation; nullability is an optional bit-packed 'Bitmap' (@Nothing@ = no+nulls, @Just bm@ = bit @i@ set iff row @i@ is valid).+-}+data Column where+ BoxedColumn :: (Columnable a) => Maybe Bitmap -> VB.Vector a -> Column+ UnboxedColumn ::+ (Columnable a, VU.Unbox a) => Maybe Bitmap -> VU.Vector a -> Column+ -- TODO: mchavinda - investigate splitting this into a separate intermediate+ -- representation.+ -- Efficient intermediate formats.z+ -- Bit-packed Text: shared UTF-8 byte buffer + row offsets + optional bitmap;+ -- Text is materialized on demand. Only CSV ingest emits this; user-built+ -- Text columns stay 'BoxedColumn'.+ PackedText :: Maybe Bitmap -> {-# UNPACK #-} !PackedTextData -> Column+ -- A join's same-named non-key column pair ('mkMergedColumns'): both sides+ -- keep their native (packed/dict/unboxed) representation; per-row 'These'+ -- values only materialize on element access ('materializeMerged').+ MergedColumn :: !Column -> !Column -> Column++instance Show Column where+ show :: Column -> String+ show c@(MergedColumn _ _) = show (materializeMerged c)+ show (BoxedColumn Nothing column) = show column+ show (BoxedColumn (Just bm) column) =+ let n = VB.length column+ elems =+ [ if bitmapTestBit bm i then show (VB.unsafeIndex column i) else "null"+ | i <- [0 .. n - 1]+ ]+ in "[" ++ foldl (\acc e -> if null acc then e else acc ++ "," ++ e) "" elems ++ "]"+ show (UnboxedColumn Nothing column) = show column+ show (UnboxedColumn (Just bm) column) =+ let n = VU.length column+ elems =+ [ if bitmapTestBit bm i then show (VU.unsafeIndex column i) else "null"+ | i <- [0 .. n - 1]+ ]+ in "[" ++ foldl (\acc e -> if null acc then e else acc ++ "," ++ e) "" elems ++ "]"+ show c@(PackedText _ _) = show (materializePacked c)++{- | Compare two nullable boxed columns element by element, skipping null slots.+Uses a manual loop to avoid stream fusion forcing null-slot error thunks.+-}+eqBoxedCols ::+ (Eq a) => Maybe Bitmap -> VB.Vector a -> Maybe Bitmap -> VB.Vector a -> Bool+eqBoxedCols bm1 a bm2 b+ | VB.length a /= VB.length b = False+ | otherwise = go 0+ where+ !n = VB.length a+ go !i+ | i >= n = True+ | nullA || nullB = (nullA == nullB) && go (i + 1)+ | VB.unsafeIndex a i == VB.unsafeIndex b i = go (i + 1)+ | otherwise = False+ where+ nullA = maybe False (\bm -> not (bitmapTestBit bm i)) bm1+ nullB = maybe False (\bm -> not (bitmapTestBit bm i)) bm2+{-# INLINE eqBoxedCols #-}++instance Eq Column where+ (==) :: Column -> Column -> Bool+ (==) (BoxedColumn bm1 (a :: VB.Vector t1)) (BoxedColumn bm2 (b :: VB.Vector t2)) =+ case testEquality (typeRep @t1) (typeRep @t2) of+ Nothing -> False+ Just Refl -> eqBoxedCols bm1 a bm2 b+ (==) (UnboxedColumn bm1 (a :: VU.Vector t1)) (UnboxedColumn bm2 (b :: VU.Vector t2)) =+ case testEquality (typeRep @t1) (typeRep @t2) of+ Nothing -> False+ Just Refl ->+ VU.length a == VU.length b+ && VU.and+ ( VU.imap+ ( \i x ->+ let nullA = maybe False (\bm -> not (bitmapTestBit bm i)) bm1+ nullB = maybe False (\bm -> not (bitmapTestBit bm i)) bm2+ in if nullA || nullB then nullA == nullB else x == VU.unsafeIndex b i+ )+ a+ )+ (==) lhs@(MergedColumn _ _) rhs = materializeMerged lhs == rhs+ (==) lhs rhs@(MergedColumn _ _) = lhs == materializeMerged rhs+ (==) (PackedText bm1 p1) (PackedText bm2 p2) = eqPackedCols bm1 p1 bm2 p2+ (==) lhs@(PackedText _ _) rhs = materializePacked lhs == rhs+ (==) lhs rhs@(PackedText _ _) = lhs == materializePacked rhs+ (==) _ _ = False++{- | Byte-slice equality of two packed-text columns, skipping null slots+(a null compares equal only to a null), mirroring 'eqBoxedCols'.+-}+eqPackedCols ::+ Maybe Bitmap -> PackedTextData -> Maybe Bitmap -> PackedTextData -> Bool+eqPackedCols bm1 p1 bm2 p2+ | packedLength p1 /= packedLength p2 = False+ | otherwise = go 0+ where+ !n = packedLength p1+ go !i+ | i >= n = True+ | nullA || nullB = (nullA == nullB) && go (i + 1)+ | otherwise =+ let (a1, o1, l1) = packedSlice p1 i+ (a2, o2, l2) = packedSlice p2 i+ in sliceEqBytes a1 o1 l1 a2 o2 l2 && go (i + 1)+ where+ nullA = maybe False (\bm -> not (bitmapTestBit bm i)) bm1+ nullB = maybe False (\bm -> not (bitmapTestBit bm i)) bm2+{-# INLINE eqPackedCols #-}++{- | A mutable companion struct to dataframe columns.++Used mostly as an intermediate structure for I/O.+-}+data MutableColumn where+ MBoxedColumn :: (Columnable a) => VBM.IOVector a -> MutableColumn+ MUnboxedColumn :: (Columnable a, VU.Unbox a) => VUM.IOVector a -> MutableColumn++{- | A wrapper around the type-erased 'Column' carrying a phantom element type,+used to type-check expressions. The phantom is not guaranteed to match the+underlying vector's type.+-}+data TypedColumn a where+ TColumn :: (Columnable a) => Column -> TypedColumn a++instance (Eq a) => Eq (TypedColumn a) where+ (==) :: (Eq a) => TypedColumn a -> TypedColumn a -> Bool+ (==) (TColumn a) (TColumn b) = a == b++instance (Show a) => Show (TypedColumn a) where+ show :: (Show a) => TypedColumn a -> String+ show (TColumn col) = show col++-- | Unwrap a 'TypedColumn' back to its type-erased 'Column'.+unwrapTypedColumn :: TypedColumn a -> Column+unwrapTypedColumn (TColumn value) = value++{- | Decode a 'PackedText' into a @BoxedColumn Text@ (bit-identical to+materializing at freeze). Identity on every other column.+-}+materializePacked :: Column -> Column+materializePacked (PackedText bm p) =+ BoxedColumn bm (VB.generate (packedLength p) (packedIndexText p))+materializePacked c = c+{-# INLINE materializePacked #-}++-- | Return the 'Maybe Bitmap' from a column.+columnBitmap :: Column -> Maybe Bitmap+columnBitmap (BoxedColumn bm _) = bm+columnBitmap (UnboxedColumn bm _) = bm+columnBitmap (PackedText bm _) = bm+columnBitmap (MergedColumn _ _) = Nothing++{- | A class for converting a vector to a column of the appropriate type.+Given each Rep we tell the `toColumnRep` function which Column type to pick.+-}+class ColumnifyRep (r :: Rep) a where+ toColumnRep :: VB.Vector a -> Column++instance+ (Columnable a, VU.Unbox a) =>+ ColumnifyRep 'RUnboxed a+ where+ toColumnRep :: (Columnable a, VUM.Unbox a) => VB.Vector a -> Column+ toColumnRep v = UnboxedColumn Nothing (VU.convert v)++instance+ (Columnable a) =>+ ColumnifyRep 'RBoxed a+ where+ toColumnRep :: (Columnable a) => VB.Vector a -> Column+ toColumnRep = BoxedColumn Nothing++instance+ (Columnable a) =>+ ColumnifyRep 'RNullableBoxed (Maybe a)+ where+ toColumnRep :: (Columnable a) => VB.Vector (Maybe a) -> Column+ toColumnRep v =+ let+ n = VB.length v+ nullIdxs = VU.filter (isNothing . VB.unsafeIndex v) (VU.enumFromN 0 n)+ bm =+ if VU.null nullIdxs then allValidBitmap n else buildBitmapFromNulls' n nullIdxs+ in+ case sUnbox @a of+ STrue -> UnboxedColumn (Just bm) $ runST $ do+ mv <- VUM.new n+ VG.iforM_ v $ \i mx -> forM_ mx (VUM.unsafeWrite mv i)+ VU.unsafeFreeze mv+ SFalse ->+ BoxedColumn+ (Just bm)+ (VB.map (fromMaybe (errorWithoutStackTrace "toColumnRep: Nothing slot")) v)++-- | O(1) Gets the number of elements in the column.+columnLength :: Column -> Int+columnLength (MergedColumn a b) = min (columnLength a) (columnLength b)+columnLength (BoxedColumn _ xs) = VB.length xs+columnLength (UnboxedColumn _ xs) = VU.length xs+columnLength (PackedText _ p) = packedLength p+{-# INLINE columnLength #-}++-- | O(n) Takes the first n values of a column.+takeColumn :: Int -> Column -> Column+takeColumn n (MergedColumn a b) = MergedColumn (takeColumn n a) (takeColumn n b)+takeColumn n (BoxedColumn bm xs) =+ BoxedColumn (fmap (bitmapSlice 0 n) bm) (VG.take n xs)+takeColumn n (UnboxedColumn bm xs) =+ UnboxedColumn (fmap (bitmapSlice 0 n) bm) (VG.take n xs)+takeColumn n (PackedText bm p) =+ PackedText (fmap (bitmapSlice 0 n) bm) (packedTake n p)+{-# INLINE takeColumn #-}++{- | Merge two columns using `These`. O(1): the sides are kept in their+native representation and 'These' values materialize on element access.+-}+mkMergedColumns :: Column -> Column -> Column+mkMergedColumns = MergedColumn+{-# INLINE mkMergedColumns #-}++-- | Decode a 'MergedColumn' into the eager @BoxedColumn (These a b)@ form.+materializeMerged :: Column -> Column+materializeMerged (MergedColumn colA colB) =+ mergeEager (materializeMerged colA) (materializeMerged colB)+materializeMerged c = c++mergedHead :: Column -> Column+mergedHead (MergedColumn a b) =+ materializeMerged (MergedColumn (takeColumn 1 a) (takeColumn 1 b))+mergedHead c = c++{- | The eager element-wise merge ('These' per row, boxed). Bitmaps are+honored for every representation pair: a null side yields 'This'/'That',+both-null is an error (the join kernels never produce such a row).+-}+mergeEager :: Column -> Column -> Column+mergeEager colA colB = case (colA, colB) of+ (MergedColumn a b, _) -> mergeEager (mergeEager a b) colB+ (_, MergedColumn a b) -> mergeEager colA (mergeEager a b)+ (PackedText _ _, _) -> mergeEager (materializePacked colA) colB+ (_, PackedText _ _) -> mergeEager colA (materializePacked colB)+ (BoxedColumn bmA c1, BoxedColumn bmB c2) ->+ merged bmA bmB (VG.length c1) (VG.length c2) (c1 VG.!) (c2 VG.!)+ (BoxedColumn bmA c1, UnboxedColumn bmB c2) ->+ merged bmA bmB (VG.length c1) (VG.length c2) (c1 VG.!) (c2 VG.!)+ (UnboxedColumn bmA c1, BoxedColumn bmB c2) ->+ merged bmA bmB (VG.length c1) (VG.length c2) (c1 VG.!) (c2 VG.!)+ (UnboxedColumn bmA c1, UnboxedColumn bmB c2) ->+ merged bmA bmB (VG.length c1) (VG.length c2) (c1 VG.!) (c2 VG.!)+ where+ merged ::+ (Columnable a, Columnable b) =>+ Maybe Bitmap ->+ Maybe Bitmap ->+ Int ->+ Int ->+ (Int -> a) ->+ (Int -> b) ->+ Column+ merged bmA bmB lenA lenB atA atB =+ BoxedColumn Nothing $ VB.generate (min lenA lenB) $ \i ->+ case (validAt bmA i, validAt bmB i) of+ (True, True) -> These (atA i) (atB i)+ (True, False) -> This (atA i)+ (False, True) -> That (atB i)+ (False, False) -> error "mkMergedColumns: both null"+ validAt mbm i = maybe True (`bitmapTestBit` i) mbm+ {-# INLINE validAt #-}
+ src-internal/DataFrame/Internal/Column/Bitmap.hs view
@@ -0,0 +1,188 @@+{-# LANGUAGE BangPatterns #-}++module DataFrame.Internal.Column.Bitmap where++import Control.Monad (foldM_, forM_, when)+import Control.Monad.ST (ST, runST)+import Data.Bits (+ complement,+ popCount,+ setBit,+ shiftL,+ shiftR,+ testBit,+ (.&.),+ (.|.),+ )+import Data.List (foldl')+import Data.Maybe (fromMaybe, isNothing)+import qualified Data.Vector.Unboxed as VU+import qualified Data.Vector.Unboxed.Mutable as VUM+import Data.Word (Word8)++-- | A bit-packed validity bitmap. Bit @i@ = 1 means row @i@ is valid (not null).+type Bitmap = VU.Vector Word8++-- | A bitmap attached to its row counts so we can splice it.+data Validity = Validity !(Maybe Bitmap) {-# UNPACK #-} !Int++vBitmap :: Validity -> Maybe Bitmap+vBitmap (Validity bm _) = bm++vRowCount :: Validity -> Int+vRowCount (Validity _ n) = n++{- | Test whether row @i@ is valid (not null) in a bitmap.++The bit-level arithmetric is dense but read this as:+`shiftR` 3 is equivalent to `div` 8, and `.&. 7` is equivalent to `mod` 8.+-}+bitmapTestBit :: Bitmap -> Int -> Bool+bitmapTestBit bm i = testBit (VU.unsafeIndex bm (i `shiftR` 3)) (i .&. 7)+{-# INLINE bitmapTestBit #-}++-- | Build a fully-valid bitmap for @n@ rows (all bits set).+allValidBitmap :: Int -> Bitmap+allValidBitmap n = runST (allValidBitmap' n >>= VU.unsafeFreeze)+{-# INLINE allValidBitmap #-}++allValidBitmap' :: Int -> ST s (VUM.MVector s Word8)+allValidBitmap' n =+ let+ bytes = (n + 7) `shiftR` 3+ lastBits = n .&. 7+ lastByte = if lastBits == 0 then 0xFF else (1 `shiftL` lastBits) - 1+ in+ if bytes == 0+ then VUM.new 0+ else do+ mv <- VUM.replicate bytes (0xFF :: Word8) :: ST s (VUM.MVector s Word8)+ when (lastBits /= 0) $ VUM.unsafeWrite mv (bytes - 1) lastByte+ pure mv+{-# INLINE allValidBitmap' #-}++{- | Build a bitmap from a @VU.Vector Word8@ validity vector+(1 = valid, 0 = null), as produced by Arrow / Parquet decoders.+-}+buildBitmapFromValid :: VU.Vector Word8 -> Bitmap+buildBitmapFromValid valid =+ let n = VU.length valid+ bytes = (n + 7) `shiftR` 3+ in VU.generate bytes $ \b ->+ let base = b `shiftL` 3+ setBitIf acc bit =+ let idx = base + bit+ in if idx < n && VU.unsafeIndex valid idx /= 0+ then setBit acc bit+ else acc+ in foldl' setBitIf (0 :: Word8) [0 .. 7]++{- | Build a bitmap from a list of null-row indices.+@nullIdxs@ are the positions that are NULL.+-}+buildBitmapFromNulls :: Int -> [Int] -> Bitmap+buildBitmapFromNulls n idxs = buildBitmapFromNulls' n (VU.fromList idxs)++buildBitmapFromNulls' :: Int -> VU.Vector Int -> VU.Vector Word8+buildBitmapFromNulls' n nullIdxs = runST $ do+ bm' <- allValidBitmap' n+ VU.forM_ nullIdxs $ \i -> do+ let byteIdx = i `shiftR` 3+ bitIdx = i .&. 7+ v <- VUM.unsafeRead bm' byteIdx+ VUM.unsafeWrite bm' byteIdx (clearBit8 v bitIdx)+ VU.unsafeFreeze bm'+ where+ clearBit8 :: Word8 -> Int -> Word8+ clearBit8 b bit = b .&. complement (1 `shiftL` bit)++-- | Slice a bitmap for rows @[start .. start+len-1]@.+bitmapSlice :: Int -> Int -> Bitmap -> Bitmap+bitmapSlice start len bm+ | start .&. 7 == 0 =+ let startByte = start `shiftR` 3+ bytes = min ((len + 7) `shiftR` 3) (VU.length bm - startByte)+ in VU.slice startByte bytes bm+ | otherwise =+ let n = min len (VU.length bm `shiftL` 3 - start)+ in buildBitmapFromValid $+ VU.generate n $+ \i -> if bitmapTestBit bm (start + i) then 1 else 0++{- | Count the set bits among the first @n@ bits of a bitmap. A bitmap does+not know the length of the column it describes, and 'bitmapSlice' keeps whole+bytes on its aligned path, so the bits past @n@ may still describe rows+outside the slice.+-}+popCountUpTo :: Int -> Bitmap -> Int+popCountUpTo n bm = whole + partial+ where+ !fullBytes = min (n `shiftR` 3) (VU.length bm)+ !rest = n .&. 7+ whole = VU.foldl' (\acc b -> acc + popCount b) 0 (VU.take fullBytes bm)+ partial+ | rest == 0 || fullBytes >= VU.length bm = 0+ | otherwise =+ popCount (VU.unsafeIndex bm fullBytes .&. ((1 `shiftL` rest) - 1))+{-# INLINE popCountUpTo #-}++-- | Concatenate two bitmaps covering @n1@ and @n2@ rows respectively.+bitmapConcat :: Int -> Bitmap -> Int -> Bitmap -> Bitmap+bitmapConcat n1 bm1 n2 bm2 =+ buildBitmapFromValid $+ VU.generate (n1 + n2) $ \i ->+ if i < n1+ then if bitmapTestBit bm1 i then 1 else 0+ else if bitmapTestBit bm2 (i - n1) then 1 else 0++-- | Combine two bitmaps with AND (both must be valid for result to be valid).+andBitmaps :: Bitmap -> Bitmap -> Bitmap+andBitmaps = VU.zipWith (.&.)++{- | Splice chunk bitmaps end to end at the bit level. 'Nothing' if no chunk+carries a bitmap; chunks without one count as all-valid otherwise.+-}+concatValidity :: [Validity] -> Maybe Bitmap+concatValidity parts+ | all (isNothing . vBitmap) parts = Nothing+ | otherwise = Just $ VU.create $ do+ let total = sum (map vRowCount parts)+ outBytes = (total + 7) `shiftR` 3+ mv <- VUM.replicate outBytes 0+ let orInto i w =+ when (i < outBytes && w /= 0) $ do+ old <- VUM.unsafeRead mv i+ VUM.unsafeWrite mv i (old .|. w)+ splice !bitPos (Validity !mb !len) = do+ let bm = fromMaybe (allValidBitmap len) mb+ sh = bitPos .&. 7+ byte0 = bitPos `shiftR` 3+ lastIdx = ((len + 7) `shiftR` 3) - 1+ tailBits = len .&. 7+ lastMask =+ if tailBits == 0 then 0xFF else (1 `shiftL` tailBits) - 1+ forM_ [0 .. lastIdx] $ \k -> do+ let raw = VU.unsafeIndex bm k+ masked = if k == lastIdx then raw .&. lastMask else raw+ w = fromIntegral masked :: Word+ orInto (byte0 + k) (fromIntegral (w `shiftL` sh))+ when (sh /= 0) $+ orInto (byte0 + k + 1) (fromIntegral (w `shiftR` (8 - sh)))+ pure (bitPos + len)+ foldM_ splice 0 parts+ pure mv++-- | Pack a 0\/1 byte-per-row validity prefix into a bit-packed 'Bitmap'.+packValidity :: Int -> VUM.MVector s Word8 -> ST s Bitmap+packValidity n val = do+ bytes <- VU.unsafeFreeze (VUM.slice 0 n val)+ let assemble b =+ let base = b `shiftL` 3+ m = min 8 (n - base)+ go !acc !k+ | k >= m = acc+ | VU.unsafeIndex bytes (base + k) /= 0 =+ go (acc .|. (1 `shiftL` k)) (k + 1)+ | otherwise = go acc (k + 1)+ in go (0 :: Word8) 0+ pure $! VU.generate ((n + 7) `shiftR` 3) assemble
+ src-internal/DataFrame/Internal/Column/Builder.hs view
@@ -0,0 +1,306 @@+{-# LANGUAGE BangPatterns #-}+{-# LANGUAGE FlexibleContexts #-}+{-# LANGUAGE FlexibleInstances #-}+{-# LANGUAGE UndecidableInstances #-}++{- | Mutable, growable column builders for high-throughput ingest. No+per-append @IORef@ traffic: hot counters live in an unboxed vector, payloads+double on demand, and validity is only materialized once a null is seen.+-}+module DataFrame.Internal.Column.Builder (+ ColumnBuilder (..),+ NumBuilder,+ IntBuilder,+ DoubleBuilder,+ TextBuilder,+ TextChunk (..),+ newIntBuilder,+ newDoubleBuilder,+ newNumBuilder,+ newTextBuilder,+ appendInt,+ appendDouble,+ appendNum,+ appendText,+ appendTextSlice,+ appendTextSliceFromPtr,+ freezeTextChunk,+ concatColumns,+ mergeTextChunks,+) where++import qualified Data.Text as T+import qualified Data.Text.Array as A+import qualified Data.Vector.Unboxed as VU+import qualified Data.Vector.Unboxed.Mutable as VUM++import Control.Monad (when)+import Control.Monad.ST (ST)+import Data.Bits (shiftR)+import Data.STRef+import Data.Text.Internal (Text (..))+import Data.Word (Word8)+import DataFrame.Internal.Column (+ Column (UnboxedColumn),+ Columnable,+ )+import DataFrame.Internal.Column.Bitmap (packValidity)+import DataFrame.Internal.Column.Merge (+ TextChunk (..),+ concatColumns,+ mergeTextChunks,+ )+import Foreign.Ptr (Ptr)++{- | Operations shared by all column builders.++NB: Do not use after freezing+-}+class ColumnBuilder b where+ -- | Append a null row (sentinel payload + invalid bit).+ appendNull :: b s -> ST s ()++ -- | Rows appended so far.+ builderLength :: b s -> ST s Int++ -- | Freeze into a fully-forced 'Column'.+ freezeBuilder :: b s -> ST s Column++-- Counter slots shared by the builders: rows, any-null flag, text bytes used.+cRows, cAnyNull, cBytes :: Int+cRows = 0+cAnyNull = 1+cBytes = 2++{- | Builder for unboxed numeric payloads ('Int', 'Double', ...). 'nbNull'+is the sentinel written into null slots.+-}+data NumBuilder a s = NumBuilder+ { nbNull :: !a+ , nbCounters :: !(VUM.MVector s Int)+ , nbArrays :: !(STRef s (NumArrays a s))+ }++data NumArrays a s = NumArrays+ { naData :: !(VUM.MVector s a)+ , naValid :: !(VUM.MVector s Word8)+ }++type IntBuilder = NumBuilder Int++type DoubleBuilder = NumBuilder Double++-- | New numeric builder with a row-capacity hint and a null sentinel.+{-# SPECIALIZE newNumBuilder :: Int -> Int -> ST s (NumBuilder Int s) #-}+{-# SPECIALIZE newNumBuilder :: Double -> Int -> ST s (NumBuilder Double s) #-}+newNumBuilder :: (VU.Unbox a) => a -> Int -> ST s (NumBuilder a s)+newNumBuilder nullValue hint = do+ let cap = max 16 hint+ counters <- VUM.replicate 2 0+ dat <- VUM.unsafeNew cap+ val <- VUM.unsafeNew cap+ NumBuilder nullValue counters <$> newSTRef (NumArrays dat val)++newIntBuilder :: Int -> ST s (IntBuilder s)+newIntBuilder = newNumBuilder 0++newDoubleBuilder :: Int -> ST s (DoubleBuilder s)+newDoubleBuilder = newNumBuilder 0++appendNum :: (VU.Unbox a) => NumBuilder a s -> a -> ST s ()+appendNum b !x = do+ n <- VUM.unsafeRead (nbCounters b) cRows+ anyNull <- VUM.unsafeRead (nbCounters b) cAnyNull+ NumArrays dat val <- reserveNum b n+ VUM.unsafeWrite dat n x+ when (anyNull /= 0) $ VUM.unsafeWrite val n 1+ VUM.unsafeWrite (nbCounters b) cRows (n + 1)+{-# INLINE appendNum #-}++appendInt :: IntBuilder s -> Int -> ST s ()+appendInt = appendNum+{-# INLINE appendInt #-}++appendDouble :: DoubleBuilder s -> Double -> ST s ()+appendDouble = appendNum+{-# INLINE appendDouble #-}++-- Fetch the arrays, growing (doubling) first if row @n@ would not fit.+reserveNum :: (VU.Unbox a) => NumBuilder a s -> Int -> ST s (NumArrays a s)+reserveNum b n = do+ arrs <- readSTRef (nbArrays b)+ if n < VUM.length (naData arrs) then pure arrs else growNum b arrs+{-# INLINE reserveNum #-}++growNum ::+ (VU.Unbox a) => NumBuilder a s -> NumArrays a s -> ST s (NumArrays a s)+growNum b (NumArrays dat val) = do+ let cap = VUM.length dat+ dat' <- VUM.unsafeGrow dat cap+ val' <- VUM.unsafeGrow val cap+ let arrs = NumArrays dat' val'+ writeSTRef (nbArrays b) arrs+ pure arrs++instance (Columnable a, VU.Unbox a) => ColumnBuilder (NumBuilder a) where+ appendNull b = do+ n <- VUM.unsafeRead (nbCounters b) cRows+ anyNull <- VUM.unsafeRead (nbCounters b) cAnyNull+ NumArrays dat val <- reserveNum b n+ VUM.unsafeWrite dat n (nbNull b)+ when (anyNull == 0) $ do+ VUM.set (VUM.slice 0 n val) 1+ VUM.unsafeWrite (nbCounters b) cAnyNull 1+ VUM.unsafeWrite val n 0+ VUM.unsafeWrite (nbCounters b) cRows (n + 1)+ {-# INLINE appendNull #-}++ builderLength b = VUM.unsafeRead (nbCounters b) cRows++ freezeBuilder b = do+ n <- VUM.unsafeRead (nbCounters b) cRows+ anyNull <- VUM.unsafeRead (nbCounters b) cAnyNull+ NumArrays dat val <- readSTRef (nbArrays b)+ !vs <- freezeTrimmed n dat+ if anyNull /= 0+ then do+ !bm <- packValidity n val+ pure $! UnboxedColumn (Just bm) vs+ else pure $! UnboxedColumn Nothing vs++-- Zero-copy freeze; copies to exact size when slack exceeds a quarter of n.+freezeTrimmed :: (VU.Unbox a) => Int -> VUM.MVector s a -> ST s (VU.Vector a)+freezeTrimmed n mv+ | VUM.length mv - n <= n `shiftR` 2 = VU.unsafeFreeze (VUM.slice 0 n mv)+ | otherwise = VU.freeze (VUM.slice 0 n mv)++{- | Builder for 'Text' columns.++Representation is packed. I.e all field bytes go into one exponentially+grown byte array with rows recorded as offsets.+-}+data TextBuilder s = TextBuilder+ { tbCounters :: !(VUM.MVector s Int)+ , tbArrays :: !(STRef s (TextArrays s))+ }++data TextArrays s = TextArrays+ { taBytes :: !(A.MArray s)+ , taByteCap :: !Int+ , taOffsets :: !(VUM.MVector s Int)+ -- ^ Row @i@ spans bytes @[offsets!i, offsets!(i+1))@.+ , taValid :: !(VUM.MVector s Word8)+ }++-- | New text builder with row-count and total-byte capacity hints.+newTextBuilder :: Int -> Int -> ST s (TextBuilder s)+newTextBuilder rowHint byteHint = do+ let rcap = max 16 rowHint+ bcap = max 64 byteHint+ counters <- VUM.replicate 3 0+ bytes <- A.new bcap+ offsets <- VUM.unsafeNew (rcap + 1)+ VUM.unsafeWrite offsets 0 0+ val <- VUM.unsafeNew rcap+ TextBuilder counters <$> newSTRef (TextArrays bytes bcap offsets val)++-- | Append @len@ raw bytes at @off@ in @src@ as one field (one memcpy).+appendTextSlice :: TextBuilder s -> A.Array -> Int -> Int -> ST s ()+appendTextSlice b src off len = do+ (n, pos, arrs) <- reserveText b len+ A.copyI len (taBytes arrs) pos src off+ finishTextAppend b arrs n (pos + len)+{-# INLINE appendTextSlice #-}++-- | 'appendTextSlice' from foreign memory (e.g. an mmapped file buffer).+appendTextSliceFromPtr :: TextBuilder s -> Ptr Word8 -> Int -> ST s ()+appendTextSliceFromPtr b ptr len = do+ (n, pos, arrs) <- reserveText b len+ A.copyFromPointer (taBytes arrs) pos ptr len+ finishTextAppend b arrs n (pos + len)+{-# INLINE appendTextSliceFromPtr #-}++-- | Append an already-decoded 'Text' (its bytes are UTF-8 already).+appendText :: TextBuilder s -> T.Text -> ST s ()+appendText b (Text src off len) = appendTextSlice b src off len+{-# INLINE appendText #-}++finishTextAppend :: TextBuilder s -> TextArrays s -> Int -> Int -> ST s ()+finishTextAppend b arrs n endPos = do+ anyNull <- VUM.unsafeRead (tbCounters b) cAnyNull+ when (anyNull /= 0) $ VUM.unsafeWrite (taValid arrs) n 1+ VUM.unsafeWrite (taOffsets arrs) (n + 1) endPos+ VUM.unsafeWrite (tbCounters b) cRows (n + 1)+ VUM.unsafeWrite (tbCounters b) cBytes endPos+{-# INLINE finishTextAppend #-}++reserveText :: TextBuilder s -> Int -> ST s (Int, Int, TextArrays s)+reserveText b extra = do+ n <- VUM.unsafeRead (tbCounters b) cRows+ pos <- VUM.unsafeRead (tbCounters b) cBytes+ arrs <- readSTRef (tbArrays b)+ arrs' <-+ if n < VUM.length (taValid arrs) && pos + extra <= taByteCap arrs+ then pure arrs+ else growText b arrs (n + 1) (pos + extra)+ pure (n, pos, arrs')+{-# INLINE reserveText #-}++growText :: TextBuilder s -> TextArrays s -> Int -> Int -> ST s (TextArrays s)+growText b (TextArrays bytes bcap offsets val) needRows needBytes = do+ let rcap = VUM.length val+ (offsets', val') <-+ if needRows > rcap+ then do+ let rcap' = max (2 * rcap) needRows+ o <- VUM.unsafeGrow offsets (rcap' - rcap)+ v <- VUM.unsafeGrow val (rcap' - rcap)+ pure (o, v)+ else pure (offsets, val)+ (bytes', bcap') <-+ if needBytes > bcap+ then do+ let cap' = max (2 * bcap) needBytes+ bs <- A.resizeM bytes cap'+ pure (bs, cap')+ else pure (bytes, bcap)+ let arrs = TextArrays bytes' bcap' offsets' val'+ writeSTRef (tbArrays b) arrs+ pure arrs++{- | Freeze a 'TextBuilder' into a raw 'TextChunk' for later merging+('mergeTextChunks').+-}+freezeTextChunk :: TextBuilder s -> ST s TextChunk+freezeTextChunk b = do+ n <- VUM.unsafeRead (tbCounters b) cRows+ anyNull <- VUM.unsafeRead (tbCounters b) cAnyNull+ used <- VUM.unsafeRead (tbCounters b) cBytes+ TextArrays bytes bcap offsets val <- readSTRef (tbArrays b)+ when (used < bcap) (A.shrinkM bytes used)+ arr <- A.unsafeFreeze bytes+ offs <- VU.unsafeFreeze (VUM.slice 0 (n + 1) offsets)+ bm <-+ if anyNull /= 0+ then Just <$> packValidity n val+ else pure Nothing+ pure (TextChunk arr used offs bm)++instance ColumnBuilder TextBuilder where+ appendNull b = do+ (n, pos, arrs) <- reserveText b 0+ anyNull <- VUM.unsafeRead (tbCounters b) cAnyNull+ when (anyNull == 0) $ do+ VUM.set (VUM.slice 0 n (taValid arrs)) 1+ VUM.unsafeWrite (tbCounters b) cAnyNull 1+ VUM.unsafeWrite (taValid arrs) n 0+ VUM.unsafeWrite (taOffsets arrs) (n + 1) pos+ VUM.unsafeWrite (tbCounters b) cRows (n + 1)+ VUM.unsafeWrite (tbCounters b) cBytes pos+ {-# INLINE appendNull #-}++ builderLength b = VUM.unsafeRead (tbCounters b) cRows++ freezeBuilder b = do+ chunk <- freezeTextChunk b+ pure $! mergeTextChunks [chunk]
+ src-internal/DataFrame/Internal/Column/Conversion.hs view
@@ -0,0 +1,441 @@+{-# LANGUAGE AllowAmbiguousTypes #-}+{-# LANGUAGE BangPatterns #-}+{-# LANGUAGE ConstraintKinds #-}+{-# LANGUAGE DataKinds #-}+{-# LANGUAGE FlexibleContexts #-}+{-# LANGUAGE FlexibleInstances #-}+{-# LANGUAGE GADTs #-}+{-# LANGUAGE MultiParamTypeClasses #-}+{-# LANGUAGE OverloadedStrings #-}+{-# LANGUAGE PolyKinds #-}+{-# LANGUAGE RankNTypes #-}+{-# LANGUAGE ScopedTypeVariables #-}+{-# LANGUAGE TypeApplications #-}+{-# LANGUAGE TypeFamilies #-}+{-# LANGUAGE UndecidableInstances #-}++{- |+Conversions between 'Column' and ordinary vectors\/lists, plus the typed+extraction functions ('toVector', 'toDoubleVector', ...) that recover a+column's element type.+-}+module DataFrame.Internal.Column.Conversion where++import qualified Data.Text as T+import qualified Data.Vector as VB+import qualified Data.Vector.Generic as VG+import qualified Data.Vector.Unboxed as VU+import qualified Data.Vector.Unboxed.Mutable as VUM++import Control.Exception (throw)+import Control.Monad.ST (ST)+import Data.Kind (Type)+import Data.Type.Equality (TestEquality (..))+import Data.Word (Word8)+import DataFrame.Errors (+ DataFrameException (ExpectedNonNullableException, TypeMismatchException),+ TypeErrorContext (+ MkTypeErrorContext,+ callingFunctionName,+ errorColumnName,+ expectedType,+ userType+ ),+ )+import DataFrame.Internal.Column.Base+import DataFrame.Internal.Column.Bitmap+import DataFrame.Internal.Column.Properties+import DataFrame.Internal.Column.Types+import DataFrame.Internal.Data.PackedText (packedIndexText, packedLength)+import System.Random (RandomGen, UniformRange, uniformR)+import Type.Reflection (TypeRep, Typeable, typeRep, type (:~:) (Refl))++{- | O(n) Convert a vector to a column. Automatically picks the best representation of a vector to store the underlying data in.++__Examples:__++@+> import qualified Data.Vector as V+> fromVector (VB.fromList [(1 :: Int), 2, 3, 4])+[1,2,3,4]+@+-}+fromVector ::+ forall a.+ (Columnable a, ColumnifyRep (KindOf a) a) =>+ VB.Vector a -> Column+fromVector = toColumnRep @(KindOf a)++{- | O(n) Convert an unboxed vector to a column. This avoids the extra conversion if you already have the data in an unboxed vector.++__Examples:__++@+> import qualified Data.Vector.Unboxed as V+> fromUnboxedVector (VB.fromList [(1 :: Int), 2, 3, 4])+[1,2,3,4]+@+-}+fromUnboxedVector ::+ forall a. (Columnable a, VU.Unbox a) => VU.Vector a -> Column+fromUnboxedVector = UnboxedColumn Nothing++{- | O(n) Convert a list to a column. Automatically picks the best representation of a vector to store the underlying data in.++__Examples:__++@+> fromList [(1 :: Int), 2, 3, 4]+[1,2,3,4]+@+-}+fromList ::+ forall a.+ (Columnable a, ColumnifyRep (KindOf a) a) =>+ [a] -> Column+fromList = toColumnRep @(KindOf a) . VB.fromList++{- | O(n) Create a column of random elements within a range.++Takes a random number generator, a length, and a lower and upper bound for the random values.++__Examples:__++@+> import System.Random (mkStdGen)+> mkRandom (mkStdGen 42) 4 0 10+[4,2,6,5]+@+-}+mkRandom ::+ (RandomGen g, Columnable a, ColumnifyRep (KindOf a) a, UniformRange a) =>+ g -> Int -> a -> a -> Column+mkRandom pureGen k lo hi = fromList $ go pureGen k+ where+ go _g 0 = []+ go g n =+ let+ (!v, !g') = uniformR (lo, hi) g+ in+ v : go g' (n - 1)++{- | O(n) Converts a column to a list. Throws an exception if the wrong type is specified.++__Examples:__++@+> column = fromList [(1 :: Int), 2, 3, 4]+> toList @Int column+[1,2,3,4]+> toList @Double column+exception: ...+@+-}+toList :: forall a. (Columnable a) => Column -> [a]+toList xs = case toVector @a xs of+ Left err -> throw err+ Right val -> VB.toList val++{- | Type-safe conversion of a column to a vector of element type @a@ (specify via+type application); 'Left' 'TypeMismatchException' when the column's type differs.++>>> toVector @Int @VU.Vector column+Right (unboxed vector of Ints)++>>> toVector @Text @VB.Vector column+Right (boxed vector of Text)+-}+toVector ::+ forall a v.+ (VG.Vector v a, Columnable a) => Column -> Either DataFrameException (v a)+toVector col = case col of+ PackedText _ _ -> toVector (materializePacked col)+ MergedColumn _ _ -> toVector (materializeMerged col)+ BoxedColumn bm (inner :: VB.Vector c) ->+ -- Check if user wants Maybe c (nullable) or c directly+ case testEquality (typeRep @a) (typeRep @c) of+ Just Refl -> Right $ VG.convert inner+ Nothing ->+ -- Try: a = Maybe c+ case testEquality (typeRep @a) (typeRep @(Maybe c)) of+ Just Refl ->+ -- Use VB.generate to avoid fusion forcing null slots+ let !n = VB.length inner+ maybeVec = case bm of+ Nothing -> VB.generate n (Just . VB.unsafeIndex inner)+ Just bitmap -> VB.generate n $ \i ->+ if bitmapTestBit bitmap i then Just (VB.unsafeIndex inner i) else Nothing+ in Right $ VG.convert maybeVec+ Nothing ->+ Left $+ TypeMismatchException+ ( MkTypeErrorContext+ { userType = Right (typeRep @a)+ , expectedType = Right (typeRep @c)+ , callingFunctionName = Just "toVector"+ , errorColumnName = Nothing+ }+ )+ UnboxedColumn bm (inner :: VU.Vector c) ->+ case testEquality (typeRep @a) (typeRep @c) of+ Just Refl -> Right $ VG.convert inner+ Nothing ->+ case testEquality (typeRep @a) (typeRep @(Maybe c)) of+ Just Refl ->+ let maybeVec = case bm of+ Nothing -> VB.generate (VU.length inner) (Just . VU.unsafeIndex inner)+ Just bitmap -> VB.generate (VU.length inner) $ \i ->+ if bitmapTestBit bitmap i then Just (VU.unsafeIndex inner i) else Nothing+ in Right $ VG.convert maybeVec+ Nothing ->+ Left $+ TypeMismatchException+ ( MkTypeErrorContext+ { userType = Right (typeRep @a)+ , expectedType = Right (typeRep @c)+ , callingFunctionName = Just "toVector"+ , errorColumnName = Nothing+ }+ )+{-# INLINEABLE toVector #-}++-- Some common types we will use for numerical computing.++{- | Convert a column to an unboxed 'Double' vector, coercing numeric types+('realToFrac' for floats, 'fromIntegral' for integrals; nulls become @NaN@).+'Left' 'TypeMismatchException' when the column is not numeric.+-}+toDoubleVector :: Column -> Either DataFrameException (VU.Vector Double)+toDoubleVector column =+ case column of+ PackedText _ _ -> toDoubleVector (materializePacked column)+ MergedColumn _ _ -> toDoubleVector (materializeMerged column)+ UnboxedColumn (Just _) _ -> Left ExpectedNonNullableException+ UnboxedColumn Nothing (f :: VU.Vector a) -> case testEquality (typeRep @a) (typeRep @Double) of+ Just Refl -> Right f+ Nothing -> case sFloating @a of+ STrue -> Right (VU.map realToFrac f)+ SFalse -> case sIntegral @a of+ STrue -> Right (VU.map fromIntegral f)+ SFalse ->+ Left $+ TypeMismatchException+ ( MkTypeErrorContext+ { userType = Right (typeRep @Double)+ , expectedType = Right (typeRep @a)+ , callingFunctionName = Just "toDoubleVector"+ , errorColumnName = Nothing+ }+ )+ BoxedColumn (Just _) (f :: VB.Vector a) -> case testEquality (typeRep @a) (typeRep @Integer) of+ Just Refl -> Left ExpectedNonNullableException+ Nothing ->+ Left $+ TypeMismatchException+ ( MkTypeErrorContext+ { userType = Right (typeRep @Double)+ , expectedType = Left (columnTypeString column) :: Either String (TypeRep ())+ , callingFunctionName = Just "toDoubleVector"+ , errorColumnName = Nothing+ }+ )+ BoxedColumn Nothing (f :: VB.Vector a) -> case testEquality (typeRep @a) (typeRep @Integer) of+ Just Refl -> Right (VB.convert $ VB.map fromIntegral f)+ Nothing ->+ Left $+ TypeMismatchException+ ( MkTypeErrorContext+ { userType = Right (typeRep @Double)+ , expectedType = Left (columnTypeString column) :: Either String (TypeRep ())+ , callingFunctionName = Just "toDoubleVector"+ , errorColumnName = Nothing+ }+ )++{- | Convert a column to an unboxed 'Float' vector, coercing numeric types (nulls+become @NaN@); 'Left' 'TypeMismatchException' when not numeric. Converting from+'Double' may lose precision.+-}+toFloatVector :: Column -> Either DataFrameException (VU.Vector Float)+toFloatVector column =+ case column of+ PackedText _ _ -> toFloatVector (materializePacked column)+ MergedColumn _ _ -> toFloatVector (materializeMerged column)+ UnboxedColumn bm (f :: VU.Vector a) -> case testEquality (typeRep @a) (typeRep @Float) of+ Just Refl -> case bm of+ Nothing -> Right f+ Just bitmap -> Right $ VU.imap (\i x -> if bitmapTestBit bitmap i then x else read "NaN") f+ Nothing -> case sFloating @a of+ STrue ->+ Right+ ( VU.imap+ ( \i x -> case bm of+ Just bitmap | not (bitmapTestBit bitmap i) -> read "NaN"+ _ -> realToFrac x+ )+ f+ )+ SFalse -> case sIntegral @a of+ STrue ->+ Right+ ( VU.imap+ ( \i x -> case bm of+ Just bitmap | not (bitmapTestBit bitmap i) -> read "NaN"+ _ -> fromIntegral x+ )+ f+ )+ SFalse ->+ Left $+ TypeMismatchException+ ( MkTypeErrorContext+ { userType = Right (typeRep @Float)+ , expectedType = Right (typeRep @a)+ , callingFunctionName = Just "toFloatVector"+ , errorColumnName = Nothing+ }+ )+ BoxedColumn bm (f :: VB.Vector a) -> case testEquality (typeRep @a) (typeRep @Integer) of+ Just Refl ->+ Right+ ( VB.convert $+ VB.imap+ ( \i x -> case bm of+ Just bitmap | not (bitmapTestBit bitmap i) -> read "NaN"+ _ -> fromIntegral x+ )+ f+ )+ Nothing ->+ Left $+ TypeMismatchException+ ( MkTypeErrorContext+ { userType = Right (typeRep @Float)+ , expectedType = Left (columnTypeString column) :: Either String (TypeRep ())+ , callingFunctionName = Just "toFloatVector"+ , errorColumnName = Nothing+ }+ )++{- | Convert a column to an unboxed 'Int' vector, coercing numeric types+(floats are 'round'ed via banker's rounding); 'Left' 'TypeMismatchException'+when the column is not numeric. Does not support nullable columns.+-}+toIntVector :: Column -> Either DataFrameException (VU.Vector Int)+toIntVector column =+ case column of+ PackedText _ _ -> toIntVector (materializePacked column)+ MergedColumn _ _ -> toIntVector (materializeMerged column)+ UnboxedColumn _ (f :: VU.Vector a) -> case testEquality (typeRep @a) (typeRep @Int) of+ Just Refl -> Right f+ Nothing -> case sFloating @a of+ STrue -> Right (VU.map (round . (realToFrac :: a -> Double)) f)+ SFalse -> case sIntegral @a of+ STrue -> Right (VU.map fromIntegral f)+ SFalse ->+ Left $+ TypeMismatchException+ ( MkTypeErrorContext+ { userType = Right (typeRep @Int)+ , expectedType = Right (typeRep @a)+ , callingFunctionName = Just "toIntVector"+ , errorColumnName = Nothing+ }+ )+ BoxedColumn _ (f :: VB.Vector a) -> case testEquality (typeRep @a) (typeRep @Integer) of+ Just Refl -> Right (VB.convert $ VB.map fromIntegral f)+ Nothing ->+ Left $+ TypeMismatchException+ ( MkTypeErrorContext+ { userType = Right (typeRep @Int)+ , expectedType = Left (columnTypeString column) :: Either String (TypeRep ())+ , callingFunctionName = Just "toIntVector"+ , errorColumnName = Nothing+ }+ )++toUnboxedVector ::+ forall a.+ (Columnable a, VU.Unbox a) => Column -> Either DataFrameException (VU.Vector a)+toUnboxedVector column =+ case column of+ UnboxedColumn _ (f :: VU.Vector b) -> case testEquality (typeRep @a) (typeRep @b) of+ Just Refl -> Right f+ Nothing ->+ Left $+ TypeMismatchException+ ( MkTypeErrorContext+ { userType = Right (typeRep @a)+ , expectedType = Right (typeRep @b)+ , callingFunctionName = Just "toUnboxedVector"+ , errorColumnName = Nothing+ }+ )+ _ ->+ Left $+ TypeMismatchException+ ( MkTypeErrorContext+ { userType = Right (typeRep @a)+ , expectedType = Left (columnTypeString column) :: Either String (TypeRep ())+ , callingFunctionName = Just "toUnboxedVector"+ , errorColumnName = Nothing+ }+ )+{-# INLINE toUnboxedVector #-}++-- Shared finaliser for the two parseUnboxedColumn* helpers. Freezes+-- the mutable data vector, and only materialises the bitmap when the+-- column actually had nulls.+{-# INLINE finalizeParseResult #-}+finalizeParseResult ::+ (VU.Unbox a) =>+ VUM.STVector s a ->+ VUM.STVector s Word8 ->+ Bool ->+ ST s (Maybe (Maybe Bitmap, VU.Vector a))+finalizeParseResult values vmask anyNull+ | anyNull = do+ vs <- VU.unsafeFreeze values+ vm <- VU.unsafeFreeze vmask+ return (Just (Just (buildBitmapFromValid vm), vs))+ | otherwise = do+ vs <- VU.unsafeFreeze values+ return (Just (Nothing, vs))++-- | Convert any Column to a vector of Text labels (one per row).+columnToTextVec :: Column -> VB.Vector T.Text+columnToTextVec c@(MergedColumn _ _) = columnToTextVec (materializeMerged c)+columnToTextVec (BoxedColumn bm (col' :: VB.Vector a)) =+ case bm of+ Nothing -> case testEquality (typeRep @a) (typeRep @T.Text) of+ Just Refl -> col'+ Nothing -> VB.map (T.pack . show) col'+ Just bitmap ->+ VB.imap+ (\i x -> if bitmapTestBit bitmap i then T.pack (show x) else "null")+ col'+columnToTextVec (UnboxedColumn bm col') =+ case bm of+ Nothing -> VB.map (T.pack . show) (VB.convert col')+ Just bitmap ->+ VB.generate (VU.length col') $ \i ->+ if bitmapTestBit bitmap i then T.pack (show (col' VU.! i)) else "null"+columnToTextVec (PackedText bm p) =+ VB.generate (packedLength p) $ \i -> case bm of+ Just bitmap | not (bitmapTestBit bitmap i) -> "null"+ _ -> packedIndexText p i++-- An internal helper for type errors+throwTypeMismatch ::+ forall (a :: Type) (b :: Type).+ (Typeable a, Typeable b) => Either DataFrameException Column+throwTypeMismatch =+ Left $+ TypeMismatchException+ MkTypeErrorContext+ { userType = Right (typeRep @b)+ , expectedType = Right (typeRep @a)+ , callingFunctionName = Nothing+ , errorColumnName = Nothing+ }
+ src-internal/DataFrame/Internal/Column/Encode.hs view
@@ -0,0 +1,185 @@+{-# LANGUAGE BangPatterns #-}+{-# LANGUAGE ExplicitNamespaces #-}+{-# LANGUAGE GADTs #-}+{-# LANGUAGE ScopedTypeVariables #-}+{-# LANGUAGE TypeApplications #-}++{- | Dictionary-encode a text (or factor) group key to dense @Int@ codes: each row+gets a first-appearance code @0..card-1@ (NULL reserved) plus the cardinality.++TODO: mchavinda - revise if this module is still necessary.+-}+module DataFrame.Internal.Column.Encode (+ dictEncodeColumn,+ dictEncodeColumnUpTo,+ dictCompactColumn,+ dictMaxCardinality,+) where++import Control.Monad (when)+import Control.Monad.ST (runST)+import qualified Data.Text as T+import qualified Data.Text.Array as A+import Data.Type.Equality (TestEquality (..), type (:~:) (Refl))+import qualified Data.Vector as V+import qualified Data.Vector.Unboxed as VU+import qualified Data.Vector.Unboxed.Mutable as VUM+import Type.Reflection (typeRep)++import DataFrame.Internal.Algorithms.Hash (+ fnvOffset,+ mixBytes,+ mixText,+ nullSalt,+ )+import DataFrame.Internal.Column (Column (..))+import DataFrame.Internal.Column.Bitmap (Bitmap, bitmapTestBit)+import DataFrame.Internal.Data.HashTable (htInsert, newHashTable)+import DataFrame.Internal.Data.PackedText (+ PackedTextData (..),+ mkOffsets,+ mkSel,+ packedLength,+ packedSlice,+ sliceEqBytes,+ )++{- | Largest distinct-value count we will dictionary-encode. Above this the codes+no longer index a reasonable direct accumulator and the encode pass is pure+overhead, so the caller keeps the plain hash group-by.+-}+dictMaxCardinality :: Int+dictMaxCardinality = 1048576++{- | Dictionary-encode a text-like column to dense first-appearance @Int@ codes,+returning @Just (codes, cardinality)@ (a NULL row gets its own reserved code).+'Nothing' for non-text columns or cardinality above 'dictMaxCardinality'.+-}+dictEncodeColumn :: Column -> Maybe (VU.Vector Int, Int)+dictEncodeColumn = dictEncodeColumnUpTo dictMaxCardinality++{- | Dictionary-encode like 'dictEncodeColumn' but bail to 'Nothing' as soon as+the distinct count would exceed @maxCard@, letting a low-cardinality probe avoid+a full high-cardinality pass.+-}+dictEncodeColumnUpTo :: Int -> Column -> Maybe (VU.Vector Int, Int)+dictEncodeColumnUpTo maxCard (PackedText bm p) = encodePacked maxCard bm p+dictEncodeColumnUpTo maxCard (BoxedColumn bm (v :: V.Vector a)) =+ case testEquality (typeRep @a) (typeRep @T.Text) of+ Just Refl -> encodeBoxedText maxCard bm v+ Nothing -> Nothing+dictEncodeColumnUpTo _ _ = Nothing++{- | Encode a packed-text column: hash each row's raw UTF-8 bytes (the grouping+'mixBytes'), re-verify byte equality on collisions, assign dense codes in+first-appearance order. A null row hashes 'nullSalt'.+-}+encodePacked ::+ Int -> Maybe Bitmap -> PackedTextData -> Maybe (VU.Vector Int, Int)+encodePacked maxCard bm p =+ let !n = packedLength p+ valid i = case bm of+ Just b -> bitmapTestBit b i+ Nothing -> True+ hashAt i =+ if valid i+ then let (arr, o, l) = packedSlice p i in mixBytes fnvOffset arr o l+ else nullSalt+ eqAt a b =+ case (valid a, valid b) of+ (True, True) ->+ let (arrA, oA, lA) = packedSlice p a+ (arrB, oB, lB) = packedSlice p b+ in sliceEqBytes arrA oA lA arrB oB lB+ (False, False) -> True+ _ -> False+ in buildCodes maxCard n hashAt eqAt++{- | Encode a boxed 'Data.Text.Text' column, mirroring 'encodePacked' but over+boxed values (used when a user-built Text column is grouped).+-}+encodeBoxedText ::+ Int -> Maybe Bitmap -> V.Vector T.Text -> Maybe (VU.Vector Int, Int)+encodeBoxedText maxCard bm v =+ let !n = V.length v+ valid i = case bm of+ Just b -> bitmapTestBit b i+ Nothing -> True+ hashAt i =+ if valid i then mixText fnvOffset (V.unsafeIndex v i) else nullSalt+ eqAt a b =+ case (valid a, valid b) of+ (True, True) -> V.unsafeIndex v a == V.unsafeIndex v b+ (False, False) -> True+ _ -> False+ in buildCodes maxCard n hashAt eqAt++{- | The shared code-assignment loop: bucket every row through an open-addressing+table on its precomputed hash, re-verify with @eqAt@ on a hit, assign dense+first-appearance codes. Bails to 'Nothing' once the distinct count exceeds @maxCard@.+-}+buildCodes ::+ Int -> Int -> (Int -> Int) -> (Int -> Int -> Bool) -> Maybe (VU.Vector Int, Int)+buildCodes maxCard n hashAt eqAt+ | n == 0 = Just (VU.empty, 0)+ | otherwise = runST $ do+ ht <- newHashTable (min n (maxCard + 1))+ codes <- VUM.new n+ let go !i !next+ | i >= n = pure (Just next)+ | next > maxCard = pure Nothing+ | otherwise = do+ let !h = hashAt i+ (code, isNew) <- htInsert ht eqAt next i h+ VUM.unsafeWrite codes i code+ go (i + 1) (if isNew then next + 1 else next)+ mres <- go 0 0+ case mres of+ Nothing -> pure Nothing+ Just card -> do+ frozen <- VU.unsafeFreeze codes+ pure (Just (frozen, card))++dictCompactColumn :: Column -> Column+dictCompactColumn col@(PackedText bm p) =+ case encodePacked dictMaxCardinality bm p of+ Just (codes, card)+ | 2 * card <= packedLength p ->+ PackedText bm (dictPacked p codes card)+ _ -> col+dictCompactColumn col = col++dictPacked :: PackedTextData -> VU.Vector Int -> Int -> PackedTextData+dictPacked p codes card = runST $ do+ let n = VU.length codes+ reps <- VUM.replicate card (-1)+ let findReps !i !remaining+ | remaining <= 0 || i >= n = pure ()+ | otherwise = do+ let c = VU.unsafeIndex codes i+ cur <- VUM.unsafeRead reps c+ if cur < 0+ then VUM.unsafeWrite reps c i >> findReps (i + 1) (remaining - 1)+ else findReps (i + 1) remaining+ findReps 0 card+ repsV <- VU.unsafeFreeze reps+ let lens = VU.map (\r -> let (_, _, l) = packedSlice p r in l) repsV+ offs = VU.scanl' (+) 0 lens+ total = VU.last offs+ marr <- A.new (max 1 total)+ let copyRep !c =+ when (c < card) $ do+ let r = VU.unsafeIndex repsV c+ (arr, o, l) = packedSlice p r+ A.copyI l marr (VU.unsafeIndex offs c) arr o+ copyRep (c + 1)+ copyRep 0+ arr <- A.unsafeFreeze marr+ pure+ ( PackedTextData+ { ptBytes = arr+ , ptOffsets = mkOffsets offs+ , ptSel = Just (mkSel card codes)+ , ptCanonicalSel = True+ }+ )
+ src-internal/DataFrame/Internal/Column/Merge.hs view
@@ -0,0 +1,138 @@+{-# LANGUAGE BangPatterns #-}+{-# LANGUAGE FlexibleContexts #-}+{-# LANGUAGE GADTs #-}+{-# LANGUAGE ScopedTypeVariables #-}+{-# LANGUAGE TypeApplications #-}++{- | Concatenation of per-chunk 'Column's (e.g. from parallel CSV chunks). Text+columns merge at the byte level via 'TextChunk' \/ 'mergeTextChunks', so no+per-chunk 'Data.Text.Text' values are ever materialized.+-}+module DataFrame.Internal.Column.Merge (+ TextChunk (..),+ concatColumns,+ mergeTextChunks,+ packedFromTextChunk,+ concatValidity,+ tcRows,+) where++import qualified Data.Text.Array as A+import qualified Data.Vector as VB+import qualified Data.Vector.Unboxed as VU+import qualified Data.Vector.Unboxed.Mutable as VUM++import Control.Monad (foldM_, forM_)+import Control.Monad.ST (runST)+import Data.Type.Equality (testEquality, (:~:) (Refl))+import DataFrame.Internal.Column (+ Column (..),+ Columnable,+ isMergedColumn,+ isPackedText,+ materializeMerged,+ materializePacked,+ )+import DataFrame.Internal.Column.Bitmap (+ Bitmap,+ Validity (Validity),+ concatValidity,+ )+import DataFrame.Internal.Data.PackedText (mkPackedContiguous)+import Type.Reflection (typeRep)++{- | A frozen text-builder chunk: raw UTF-8 bytes plus row offsets (row @i@+spans bytes @[offsets!i, offsets!(i+1))@) and an optional validity bitmap.+'Data.Text.Text' values are only created when chunks merge into a 'Column'.+-}+data TextChunk = TextChunk+ { tcBytes :: !A.Array+ , tcUsed :: !Int+ , tcOffsets :: !(VU.Vector Int)+ , tcBitmap :: !(Maybe Bitmap)+ }++tcRows :: TextChunk -> Int+tcRows c = VU.length (tcOffsets c) - 1++{- | Freeze a builder chunk directly into a packed-text column: no+'Data.Text.Text' materialization, no UTF-8 validation pass (deferred to decode).+Not yet called by any reader.+-}+packedFromTextChunk :: TextChunk -> Column+packedFromTextChunk (TextChunk arr _used offs bm) =+ PackedText bm (mkPackedContiguous arr offs)++{- | Merge text chunks into one packed-text 'Column': one byte-array copy per+chunk, one offset rebase, then wrap the shared buffer + offsets as 'PackedText'+(no per-row header, decode deferred).+-}+mergeTextChunks :: [TextChunk] -> Column+mergeTextChunks [] = error "DataFrame.Internal.ColumnMerge.mergeTextChunks: empty list"+mergeTextChunks [c] = packedFromTextChunk c+mergeTextChunks cs = runST $ do+ let totalBytes = sum (map tcUsed cs)+ totalRows = sum (map tcRows cs)+ arr <- A.new (max 1 totalBytes)+ offs <- VUM.unsafeNew (totalRows + 1)+ VUM.unsafeWrite offs 0 0+ let splice !byteBase !rowBase c = do+ let n = tcRows c+ co = tcOffsets c+ A.copyI (tcUsed c) arr byteBase (tcBytes c) 0+ forM_ [1 .. n] $ \i ->+ VUM.unsafeWrite offs (rowBase + i) (byteBase + VU.unsafeIndex co i)+ pure (byteBase + tcUsed c, rowBase + n)+ foldM_ (\(b, r) c -> splice b r c) (0, 0) cs+ farr <- A.unsafeFreeze arr+ foffs <- VU.unsafeFreeze offs+ let !bm = concatValidity [Validity (tcBitmap c) (tcRows c) | c <- cs]+ pure (PackedText bm (mkPackedContiguous farr foffs))++{- | Merge per-chunk columns into one column.++TODO: mchavinda - this is very similar to mappendColumns can could possibly+be defined in terms of it but I'll have to ivnestigate further.+-}+concatColumns :: [Column] -> Column+concatColumns [] = error "DataFrame.Internal.Column.Builder.concatColumns: empty list"+concatColumns [c] = c+-- Normalize on the whole list, not the head: a packed or merged chunk in any+-- position must demote every chunk to the common boxed form.+concatColumns cols@(c0 : _)+ | any isMergedColumn cols = concatColumns (map materializeMerged cols)+ | any isPackedText cols = concatColumns (map materializePacked cols)+concatColumns cols@(c0 : _) = case c0 of+ PackedText _ _ -> concatColumns (map materializePacked cols)+ MergedColumn _ _ -> concatColumns (map materializeMerged cols)+ UnboxedColumn _ (_ :: VU.Vector a) ->+ let parts = map (unboxedPart @a) cols+ !merged = VU.concat (map snd parts)+ !bm = concatValidity [Validity mb (VU.length v) | (mb, v) <- parts]+ in UnboxedColumn bm merged+ BoxedColumn _ (_ :: VB.Vector a) ->+ let parts = map (boxedPart @a) cols+ !merged = VB.concat (map snd parts)+ !bm = concatValidity [Validity mb (VB.length v) | (mb, v) <- parts]+ in BoxedColumn bm merged++unboxedPart ::+ forall a. (Columnable a, VU.Unbox a) => Column -> (Maybe Bitmap, VU.Vector a)+unboxedPart (UnboxedColumn mb (v :: VU.Vector b)) =+ case testEquality (typeRep @a) (typeRep @b) of+ Just Refl -> (mb, v)+ Nothing -> mergeMismatch+unboxedPart _ = mergeMismatch++boxedPart ::+ forall a. (Columnable a) => Column -> (Maybe Bitmap, VB.Vector a)+boxedPart (BoxedColumn mb (v :: VB.Vector b)) =+ case testEquality (typeRep @a) (typeRep @b) of+ Just Refl -> (mb, v)+ Nothing -> mergeMismatch+boxedPart _ = mergeMismatch++mergeMismatch :: a+mergeMismatch =+ error+ "DataFrame.Internal.Column.Builder.concatColumns: chunk column types differ"
+ src-internal/DataFrame/Internal/Column/Operations.hs view
@@ -0,0 +1,1369 @@+{-# LANGUAGE AllowAmbiguousTypes #-}+{-# LANGUAGE BangPatterns #-}+{-# LANGUAGE ConstraintKinds #-}+{-# LANGUAGE DataKinds #-}+{-# LANGUAGE FlexibleContexts #-}+{-# LANGUAGE FlexibleInstances #-}+{-# LANGUAGE GADTs #-}+{-# LANGUAGE LambdaCase #-}+{-# LANGUAGE MultiParamTypeClasses #-}+{-# LANGUAGE OverloadedStrings #-}+{-# LANGUAGE PolyKinds #-}+{-# LANGUAGE RankNTypes #-}+{-# LANGUAGE ScopedTypeVariables #-}+{-# LANGUAGE TypeApplications #-}+{-# LANGUAGE TypeFamilies #-}+{-# LANGUAGE UndecidableInstances #-}++{- |+Bulk operations over columns: mapping, folding, slicing, gathering, zipping,+appending, and the mutable-column IO helpers.+-}+module DataFrame.Internal.Column.Operations where++import qualified Data.Text as T+import qualified Data.Vector as VB+import qualified Data.Vector.Generic as VG+import qualified Data.Vector.Mutable as VBM+import qualified Data.Vector.Unboxed as VU+import qualified Data.Vector.Unboxed.Mutable as VUM++import Control.Monad (when)+import Control.Monad.ST (runST)+import Data.Bits (setBit, shiftL, shiftR)+import Data.Int (Int32)+import Data.Kind (Type)+import Data.Maybe (catMaybes, fromMaybe, isNothing)+import Data.Type.Equality (TestEquality (..))+import Data.Word (Word8)+import DataFrame.Errors (+ DataFrameException (EmptyDataSetException, TypeMismatchException),+ TypeErrorContext (+ MkTypeErrorContext,+ callingFunctionName,+ errorColumnName,+ expectedType,+ userType+ ),+ )+import DataFrame.Internal.Column.Base+import DataFrame.Internal.Column.Bitmap+import DataFrame.Internal.Column.Conversion+import DataFrame.Internal.Column.Properties+import DataFrame.Internal.Column.Types+import DataFrame.Internal.Control.Concurrent (+ parThreshold,+ parallelChunks_,+ shouldParallelize,+ )+import DataFrame.Internal.Data.PackedText (+ PackedOffsets,+ PackedSel (..),+ PackedTextData (..),+ offCount,+ packedGather,+ packedLength,+ )+import System.IO.Unsafe (unsafePerformIO)+import Type.Reflection (+ TypeRep,+ Typeable,+ typeOf,+ typeRep,+ withTypeable,+ type (:~:) (Refl),+ )++{- | Force evaluation of all elements in a column. Replacement for the removed+@instance NFData Column@; used by the IO and lazy-executor strict paths.+-}+forceColumn :: Column -> ()+forceColumn (BoxedColumn Nothing (v :: VB.Vector a)) = VB.foldl' (const (`seq` ())) () v+forceColumn (BoxedColumn (Just bm) (v :: VB.Vector a)) =+ let n = VB.length v+ go !i+ | i >= n = ()+ | bitmapTestBit bm i = VB.unsafeIndex v i `seq` go (i + 1)+ | otherwise = go (i + 1)+ in go 0+forceColumn (UnboxedColumn _ v) = v `seq` ()+forceColumn (PackedText _ (PackedTextData arr offs sel _)) = arr `seq` offs `seq` sel `seq` ()+forceColumn (MergedColumn a b) =+ forceColumn a `seq` forceColumn b `seq` checkMergedNoBothNull a b++{- | 'MergedColumn' defers element construction, so forcing must still surface+the one deferred error — a row null on both sides — inside strict IO/executor+boundaries. O(rows) bitmap walk, no allocation; both-null needs a bitmap on+each side, so anything else passes immediately.+-}+checkMergedNoBothNull :: Column -> Column -> ()+checkMergedNoBothNull a b = case (columnBitmap a, columnBitmap b) of+ (Just ba, Just bb) ->+ let !n = min (columnLength a) (columnLength b)+ go !i+ | i >= n = ()+ | bitmapTestBit ba i || bitmapTestBit bb i = go (i + 1)+ | otherwise = error "mkMergedColumns: both null"+ in go 0+ _ -> ()++-- | Allocate a mutable column of size @n@ matching the constructor/type of the given column.+newMutableColumn :: Int -> Column -> IO MutableColumn+newMutableColumn n (BoxedColumn _ (_ :: VB.Vector a)) =+ MBoxedColumn <$> (VBM.new n :: IO (VBM.IOVector a))+newMutableColumn n (UnboxedColumn _ (_ :: VU.Vector a)) =+ MUnboxedColumn <$> (VUM.new n :: IO (VUM.IOVector a))+newMutableColumn n c@(PackedText _ _) = newMutableColumn n (materializePacked c)+newMutableColumn n c@(MergedColumn _ _) = newMutableColumn n (materializeMerged c)++-- | Copy a column chunk into a mutable column starting at offset @off@.+copyIntoMutableColumn :: MutableColumn -> Int -> Column -> IO ()+copyIntoMutableColumn mv off c@(MergedColumn _ _) =+ copyIntoMutableColumn mv off (materializeMerged c)+copyIntoMutableColumn (MBoxedColumn (mv :: VBM.IOVector b)) off (BoxedColumn _ (v :: VB.Vector a)) =+ case testEquality (typeRep @a) (typeRep @b) of+ Just Refl -> VG.imapM_ (\i x -> VBM.unsafeWrite mv (off + i) x) v+ Nothing -> error "copyIntoMutableColumn: Boxed type mismatch"+copyIntoMutableColumn (MUnboxedColumn (mv :: VUM.IOVector b)) off (UnboxedColumn _ (v :: VU.Vector a)) =+ case testEquality (typeRep @a) (typeRep @b) of+ Just Refl -> VG.imapM_ (\i x -> VUM.unsafeWrite mv (off + i) x) v+ Nothing -> error "copyIntoMutableColumn: Unboxed type mismatch"+copyIntoMutableColumn mc off c@(PackedText _ _) =+ copyIntoMutableColumn mc off (materializePacked c)+copyIntoMutableColumn _ _ _ =+ error "copyIntoMutableColumn: constructor mismatch"++-- | Freeze a mutable column into an immutable column.+freezeMutableColumn :: MutableColumn -> IO Column+freezeMutableColumn (MBoxedColumn mv) = BoxedColumn Nothing <$> VB.unsafeFreeze mv+freezeMutableColumn (MUnboxedColumn mv) = UnboxedColumn Nothing <$> VU.unsafeFreeze mv++-- | An internal function to map a function over the values of a column.+mapColumn ::+ forall b c.+ (Columnable b, Columnable c) =>+ (b -> c) -> Column -> Either DataFrameException Column+mapColumn f = \case+ BoxedColumn bm (col :: VB.Vector a) -> runBoxed bm col+ UnboxedColumn bm (col :: VU.Vector a) -> runUnboxed bm col+ c@(PackedText _ _) -> mapColumn f (materializePacked c)+ c@(MergedColumn _ _) -> mapColumn f (materializeMerged c)+ where+ runBoxed ::+ forall a.+ (Columnable a) =>+ Maybe Bitmap -> VB.Vector a -> Either DataFrameException Column+ runBoxed bm col = case testEquality (typeRep @b) (typeRep @(Maybe a)) of+ Just Refl ->+ let !n = VB.length col+ in Right $ case sUnbox @c of+ STrue -> UnboxedColumn Nothing $+ parGenerateUnboxed n $ \i ->+ f+ ( if maybe True (`bitmapTestBit` i) bm+ then Just (VB.unsafeIndex col i)+ else Nothing+ )+ SFalse -> fromVector @c $+ VB.generate n $ \i ->+ f+ ( if maybe True (`bitmapTestBit` i) bm+ then Just (VB.unsafeIndex col i)+ else Nothing+ )+ Nothing -> case testEquality (typeRep @a) (typeRep @b) of+ Just Refl ->+ Right $ case sUnbox @c of+ STrue ->+ UnboxedColumn+ bm+ (parGenerateUnboxed (VB.length col) (f . VB.unsafeIndex col))+ SFalse -> case bm of+ Nothing -> fromVector @c (VB.map f col)+ Just _ -> BoxedColumn bm (VB.map f col)+ Nothing -> throwTypeMismatch @a @b++ runUnboxed ::+ forall a.+ (Columnable a, VU.Unbox a) =>+ Maybe Bitmap -> VU.Vector a -> Either DataFrameException Column+ runUnboxed bm col = case testEquality (typeRep @b) (typeRep @(Maybe a)) of+ Just Refl ->+ let !n = VU.length col+ in Right $ case sUnbox @c of+ STrue -> UnboxedColumn Nothing $+ parGenerateUnboxed n $ \i ->+ f+ ( if maybe True (`bitmapTestBit` i) bm+ then Just (VU.unsafeIndex col i)+ else Nothing+ )+ SFalse -> fromVector @c $+ VB.generate n $ \i ->+ f+ ( if maybe True (`bitmapTestBit` i) bm+ then Just (VU.unsafeIndex col i)+ else Nothing+ )+ Nothing -> case testEquality (typeRep @a) (typeRep @b) of+ Just Refl -> Right $ case sUnbox @c of+ STrue ->+ UnboxedColumn+ bm+ (parGenerateUnboxed (VU.length col) (f . VU.unsafeIndex col))+ SFalse -> case bm of+ Nothing -> fromVector @c (VB.generate (VU.length col) (f . VU.unsafeIndex col))+ Just _ -> BoxedColumn bm (VB.generate (VU.length col) (f . VU.unsafeIndex col))+ Nothing -> throwTypeMismatch @a @b+{-# INLINEABLE mapColumn #-}++-- | Applies a function that returns an unboxed result to an unboxed vector, storing the result in a column.+imapColumn ::+ forall b c.+ (Columnable b, Columnable c) =>+ (Int -> b -> c) -> Column -> Either DataFrameException Column+imapColumn f = \case+ BoxedColumn bm (col :: VB.Vector a) -> runBoxed bm col+ UnboxedColumn bm (col :: VU.Vector a) -> runUnboxed bm col+ c@(PackedText _ _) -> imapColumn f (materializePacked c)+ c@(MergedColumn _ _) -> imapColumn f (materializeMerged c)+ where+ runBoxed ::+ forall a.+ (Columnable a) =>+ Maybe Bitmap -> VB.Vector a -> Either DataFrameException Column+ runBoxed bm col = case testEquality (typeRep @a) (typeRep @b) of+ Just Refl -> Right $ case sUnbox @c of+ STrue ->+ UnboxedColumn+ bm+ (VU.generate (VB.length col) (\i -> f i (VB.unsafeIndex col i)))+ SFalse -> BoxedColumn bm (VB.imap f col)+ Nothing -> throwTypeMismatch @a @b++ runUnboxed ::+ forall a.+ (Columnable a, VU.Unbox a) =>+ Maybe Bitmap -> VU.Vector a -> Either DataFrameException Column+ runUnboxed bm col = case testEquality (typeRep @a) (typeRep @b) of+ Just Refl -> Right $ case sUnbox @c of+ STrue -> UnboxedColumn bm (VU.imap f col)+ SFalse -> BoxedColumn bm (VB.imap f (VG.convert col))+ Nothing -> throwTypeMismatch @a @b++-- | O(n) Takes the last n values of a column.+takeLastColumn :: Int -> Column -> Column+takeLastColumn n column = sliceColumn (columnLength column - n) n column+{-# INLINE takeLastColumn #-}++-- | O(n) Takes n values after a given column index.+sliceColumn :: Int -> Int -> Column -> Column+sliceColumn start n (MergedColumn a b) =+ MergedColumn (sliceColumn start n a) (sliceColumn start n b)+sliceColumn start n (BoxedColumn bm xs) =+ BoxedColumn (fmap (bitmapSlice start n) bm) (VG.slice start n xs)+sliceColumn start n (UnboxedColumn bm xs) =+ UnboxedColumn (fmap (bitmapSlice start n) bm) (VG.slice start n xs)+sliceColumn start n (PackedText bm p)+ -- packedGather decodes an out-of-range index as the empty string, where+ -- the boxed and unboxed arms reject the slice, so check first.+ | start < 0 || n < 0 || start + n > packedLength p =+ errorWithoutStackTrace+ ( "sliceColumn: invalid slice ("+ ++ show start+ ++ ","+ ++ show n+ ++ ","+ ++ show (packedLength p)+ ++ ")"+ )+ | otherwise =+ PackedText+ (fmap (bitmapSlice start n) bm)+ (packedGather (VU.enumFromN start n) p)+{-# INLINE sliceColumn #-}++-- | O(n) Selects the elements at a given set of indices. Does not change the order.++-------------------------------------------------------------------------------+-- Parallel element-wise kernels+-------------------------------------------------------------------------------++{- | Parallel unboxed 'VU.generate': splits the index space into one contiguous+chunk per capability, evaluates each chunk into its disjoint slice of a single+pre-allocated mutable vector, then freezes. Element @i@ depends only on @f i@,+so the result is bit-identical to the sequential 'VU.generate' regardless of+capability count. Falls back to 'VU.generate' below 'parThreshold'.+-}+{-# SPECIALIZE parGenerateUnboxed ::+ Int -> (Int -> Double) -> VU.Vector Double+ #-}+{-# SPECIALIZE parGenerateUnboxed ::+ Int -> (Int -> Float) -> VU.Vector Float+ #-}+{-# SPECIALIZE parGenerateUnboxed :: Int -> (Int -> Int) -> VU.Vector Int #-}+{-# SPECIALIZE parGenerateUnboxed :: Int -> (Int -> Bool) -> VU.Vector Bool #-}+parGenerateUnboxed :: (VU.Unbox c) => Int -> (Int -> c) -> VU.Vector c+parGenerateUnboxed n f+ | not (shouldParallelize parThreshold n) = VU.generate n f+ | otherwise = unsafePerformIO $ do+ mv <- VUM.unsafeNew n+ parallelChunks_ parThreshold n (fillGenerate mv f)+ VU.unsafeFreeze mv+{-# NOINLINE parGenerateUnboxed #-}++{- | The chunk body shared by 'parGenerateUnboxed' and+'parGenerateUnboxedInline'. INLINE so the latter's monomorphic wrappers each+get their own copy with @f@ inlined.+-}+fillGenerate ::+ (VU.Unbox c) => VUM.IOVector c -> (Int -> c) -> Int -> Int -> IO ()+fillGenerate mv f !lo !hi =+ let go !i+ | i >= hi = pure ()+ | otherwise = VUM.unsafeWrite mv i (f i) >> go (i + 1)+ in go lo+{-# INLINE fillGenerate #-}++{- | Parallel unboxed gather: element @i@ of the result is+@v ! (ix ! i)@ (unsafe indexing — callers pass in-bounds index vectors, e.g.+grouping-produced representative rows). Same chunking as 'parGenerateUnboxed'+(one contiguous chunk per capability into disjoint slices of one buffer), so+the result is bit-identical to the sequential backpermute at any capability+count; falls back to a sequential loop below 'parThreshold'.+-}+parBackpermuteUnboxed ::+ (VU.Unbox a) => VU.Vector a -> VU.Vector Int -> VU.Vector a+parBackpermuteUnboxed v ix =+ parGenerateUnboxed (VU.length ix) (VU.unsafeIndex v . VU.unsafeIndex ix)+{-# INLINE parBackpermuteUnboxed #-}++{- | 'parGenerateUnboxed' with an INLINE body: each monomorphic NOINLINE+wrapper below gets its own copy of the fill loop with @f@ inlined, so the+per-element unknown closure call (and the boxed result it returns) disappears+— on a 1e8-row gather that call+alloc dominated the whole pass. Same chunking,+bit-identical results; wrappers must stay NOINLINE so the 'unsafePerformIO'+runs once per call.+-}+parGenerateUnboxedInline :: (VU.Unbox c) => Int -> (Int -> c) -> VU.Vector c+parGenerateUnboxedInline n f+ | not (shouldParallelize parThreshold n) = VU.generate n f+ | otherwise = unsafePerformIO $ do+ mv <- VUM.unsafeNew n+ parallelChunks_ parThreshold n (fillGenerate mv f)+ VU.unsafeFreeze mv+{-# INLINE parGenerateUnboxedInline #-}++-- | Closure-free parallel 'Int' gather: @out!i = v ! (ix!i)@.+parBackpermuteInt :: VU.Vector Int -> VU.Vector Int -> VU.Vector Int+parBackpermuteInt v ix =+ parGenerateUnboxedInline+ (VU.length ix)+ (VU.unsafeIndex v . VU.unsafeIndex ix)+{-# NOINLINE parBackpermuteInt #-}++-- | Closure-free parallel 'Double' gather: @out!i = v ! (ix!i)@.+parBackpermuteDouble :: VU.Vector Double -> VU.Vector Int -> VU.Vector Double+parBackpermuteDouble v ix =+ parGenerateUnboxedInline+ (VU.length ix)+ (VU.unsafeIndex v . VU.unsafeIndex ix)+{-# NOINLINE parBackpermuteDouble #-}++{- | Closure-free double-indirection gather: @out!g = vis ! (offs!g)@ over+@length offs - 1@ groups (the representative-row build of the 'Grouped'+pattern).+-}+parBackpermute2Int :: VU.Vector Int -> VU.Vector Int -> VU.Vector Int+parBackpermute2Int vis offs =+ parGenerateUnboxedInline+ (max 0 (VU.length offs - 1))+ (VU.unsafeIndex vis . VU.unsafeIndex offs)+{-# NOINLINE parBackpermute2Int #-}++{- | Parallel boxed gather. The read side uses 'VB.unsafeIndexM' so the array+slot is fetched eagerly (element pointers are shared, elements themselves stay+un-forced, exactly as the sequential 'VB.generate' gather). Bit-identical+element values; sequential below 'parThreshold'.+-}+parBackpermuteBoxed :: VB.Vector a -> VU.Vector Int -> VB.Vector a+parBackpermuteBoxed v ix+ | not (shouldParallelize parThreshold n) =+ VB.generate n ((v `VB.unsafeIndex`) . (ix `VU.unsafeIndex`))+ | otherwise = unsafePerformIO $ do+ mv <- VBM.unsafeNew n+ parallelChunks_ parThreshold n $ \ !lo !hi ->+ let go !i+ | i >= hi = pure ()+ | otherwise = do+ x <- VB.unsafeIndexM v (VU.unsafeIndex ix i)+ VBM.unsafeWrite mv i x+ go (i + 1)+ in go lo+ VB.unsafeFreeze mv+ where+ !n = VU.length ix+{-# NOINLINE parBackpermuteBoxed #-}++{- | Clamp sentinel (negative) indices to 0 in one closure-free parallel pass.+The clamped rows read row 0's value; callers mask them via the sentinel bitmap.+-}+parClampNonNeg :: VU.Vector Int -> VU.Vector Int+parClampNonNeg ix =+ parGenerateUnboxedInline+ (VU.length ix)+ (\i -> let !x = VU.unsafeIndex ix i in max x 0)+{-# NOINLINE parClampNonNeg #-}++{- | Validity bitmap from sentinel indices (bit @i@ valid iff @ix!i >= 0@),+built one byte (8 rows) per element in parallel.+-}+parBitmapNonNeg :: VU.Vector Int -> Bitmap+parBitmapNonNeg ix =+ let !n = VU.length ix+ !nBytes = (n + 7) `shiftR` 3+ in parGenerateUnboxedInline nBytes $ \b ->+ let !base = b `shiftL` 3+ go !acc !bit+ | bit >= 8 = acc+ | otherwise =+ let !idx = base + bit+ !acc' =+ if idx < n && VU.unsafeIndex ix idx >= 0+ then setBit acc bit+ else acc+ in go acc' (bit + 1)+ in go (0 :: Word8) 0+{-# NOINLINE parBitmapNonNeg #-}++atIndicesStable :: VU.Vector Int -> Column -> Column+atIndicesStable indexes (BoxedColumn bm column) =+ BoxedColumn+ ( fmap+ ( \bm0 ->+ buildBitmapFromValid $+ VU.map (\i -> if bitmapTestBit bm0 i then 1 else 0) indexes+ )+ bm+ )+ (parBackpermuteBoxed column indexes)+atIndicesStable indexes (UnboxedColumn bm (column :: VU.Vector a)) =+ UnboxedColumn+ ( fmap+ ( \bm0 ->+ buildBitmapFromValid $+ VU.map (\i -> if bitmapTestBit bm0 i then 1 else 0) indexes+ )+ bm+ )+ -- Int/Double hit the closure-free monomorphic kernels; anything else+ -- keeps the generic (per-element closure call) path.+ ( case testEquality (typeRep @a) (typeRep @Int) of+ Just Refl -> parBackpermuteInt column indexes+ Nothing -> case testEquality (typeRep @a) (typeRep @Double) of+ Just Refl -> parBackpermuteDouble column indexes+ Nothing -> parBackpermuteUnboxed column indexes+ )+atIndicesStable indexes (MergedColumn a b) =+ MergedColumn (atIndicesStable indexes a) (atIndicesStable indexes b)+atIndicesStable indexes (PackedText bm p) =+ PackedText+ ( fmap+ ( \bm0 ->+ buildBitmapFromValid $+ VU.map (\i -> if bitmapTestBit bm0 i then 1 else 0) indexes+ )+ bm+ )+ (packedGather indexes p)+{-# INLINE atIndicesStable #-}++{- | Like 'atIndicesStable' but treats negative indices as null.+Keeps the index vector fully unboxed (no @VB.Vector (Maybe Int)@).+-}+gatherWithSentinel :: VU.Vector Int -> Column -> Column+gatherWithSentinel indices c@(MergedColumn _ _) =+ gatherWithSentinel indices (materializeMerged c)+gatherWithSentinel indices col =+ let !n = VU.length indices+ !newBm = parBitmapNonNeg indices+ withBm srcBm = case srcBm of+ Nothing -> Just newBm+ Just sb -> Just (andBitmaps newBm (gatherSrcBm sb))+ -- Sequential fallback: gather an existing source bitmap through the+ -- raw indices (negative-guarded). Only runs when the source side is+ -- itself nullable, which join build sides normally are not.+ gatherSrcBm sb =+ buildBitmapFromValid $ VU.generate n $ \i ->+ let idx = VU.unsafeIndex indices i+ in if idx >= 0 && bitmapTestBit sb idx then 1 else 0+ in case col of+ -- packedGather composes -1 sentinels into the selector natively+ -- (and shares the byte buffers), so text takes the raw indices.+ PackedText srcBm p -> PackedText (withBm srcBm) (packedGather indices p)+ BoxedColumn srcBm v+ -- An empty source means every index is a sentinel; clamping+ -- would read row 0 of an empty vector.+ | VB.null v -> BoxedColumn (withBm srcBm) (allNullBoxed n v)+ | otherwise ->+ BoxedColumn+ (withBm srcBm)+ (parBackpermuteBoxed v (parClampNonNeg indices))+ UnboxedColumn srcBm v+ | VU.null v -> UnboxedColumn (withBm srcBm) (allNullUnboxed n v)+ | otherwise ->+ -- Reuse atIndicesStable's Int/Double monomorphic kernel+ -- dispatch for the payload; the bitmap is replaced below.+ case atIndicesStable (parClampNonNeg indices) (UnboxedColumn Nothing v) of+ UnboxedColumn _ dat -> UnboxedColumn (withBm srcBm) dat+ other -> other+{-# INLINE gatherWithSentinel #-}++{- | An @n@-row payload for a gather whose source is empty: every index is a+sentinel, so the sentinel bitmap masks every row and no element is ever read.+The source vector is passed only to fix the element type.+-}+allNullBoxed :: Int -> VB.Vector a -> VB.Vector a+allNullBoxed n _ = VB.replicate n (error "gatherWithSentinel: null row forced")++-- | 'allNullBoxed' for unboxed payloads; the buffer is left uninitialised.+allNullUnboxed :: (VU.Unbox a) => Int -> VU.Vector a -> VU.Vector a+allNullUnboxed n _ = runST (VUM.new n >>= VU.unsafeFreeze)++-- | Internal helper to get indices in a boxed vector.+getIndices :: VU.Vector Int -> VB.Vector a -> VB.Vector a+getIndices indices xs = VB.generate (VU.length indices) (\i -> xs VB.! (indices VU.! i))+{-# INLINE getIndices #-}++-- | Internal helper to get indices in an unboxed vector.+getIndicesUnboxed :: (VU.Unbox a) => VU.Vector Int -> VU.Vector a -> VU.Vector a+getIndicesUnboxed indices xs = VU.generate (VU.length indices) (\i -> xs VU.! (indices VU.! i))+{-# INLINE getIndicesUnboxed #-}++findIndices ::+ forall a.+ (Columnable a) =>+ (a -> Bool) ->+ Column ->+ Either DataFrameException (VU.Vector Int)+findIndices predicate = \case+ BoxedColumn _ (v :: VB.Vector b) -> run v VG.convert+ UnboxedColumn _ (v :: VU.Vector b) -> run v id+ c@(PackedText _ _) -> findIndices predicate (materializePacked c)+ c@(MergedColumn _ _) -> findIndices predicate (materializeMerged c)+ where+ run ::+ forall b v.+ (Typeable b, VG.Vector v b, VG.Vector v Int) =>+ v b ->+ (v Int -> VU.Vector Int) ->+ Either DataFrameException (VU.Vector Int)+ run column finalize = case testEquality (typeRep @a) (typeRep @b) of+ Just Refl -> Right . finalize $ VG.findIndices predicate column+ Nothing ->+ Left $+ TypeMismatchException+ MkTypeErrorContext+ { userType = Right (typeRep @a)+ , expectedType = Right (typeRep @b)+ , callingFunctionName = Just "findIndices"+ , errorColumnName = Nothing+ }++-- | Fold (right) column with index.+ifoldrColumn ::+ forall a b.+ (Columnable a, Columnable b) =>+ (Int -> a -> b -> b) -> b -> Column -> Either DataFrameException b+ifoldrColumn f acc = \case+ BoxedColumn _ column -> foldrWorker column+ UnboxedColumn _ column -> foldrWorker column+ c@(PackedText _ _) -> ifoldrColumn f acc (materializePacked c)+ c@(MergedColumn _ _) -> ifoldrColumn f acc (materializeMerged c)+ where+ foldrWorker ::+ forall c v.+ (Typeable c, VG.Vector v c) =>+ v c ->+ Either DataFrameException b+ foldrWorker vec = case testEquality (typeRep @a) (typeRep @c) of+ Just Refl -> pure $ VG.ifoldr f acc vec+ Nothing ->+ Left $+ TypeMismatchException+ ( MkTypeErrorContext+ { userType = Right (typeRep @a)+ , expectedType = Right (typeRep @c)+ , callingFunctionName = Just "ifoldrColumn"+ , errorColumnName = Nothing+ }+ )++foldlColumn ::+ forall a b.+ (Columnable a, Columnable b) =>+ (b -> a -> b) -> b -> Column -> Either DataFrameException b+foldlColumn f acc = \case+ BoxedColumn _ column -> foldlWorker column+ UnboxedColumn _ column -> foldlWorker column+ c@(PackedText _ _) -> foldlColumn f acc (materializePacked c)+ c@(MergedColumn _ _) -> foldlColumn f acc (materializeMerged c)+ where+ foldlWorker ::+ forall c v.+ (Typeable c, VG.Vector v c) =>+ v c ->+ Either DataFrameException b+ foldlWorker vec = case testEquality (typeRep @a) (typeRep @c) of+ Just Refl -> pure $ VG.foldl' f acc vec+ Nothing ->+ Left $+ TypeMismatchException+ ( MkTypeErrorContext+ { userType = Right (typeRep @a)+ , expectedType = Right (typeRep @c)+ , callingFunctionName = Just "ifoldrColumn"+ , errorColumnName = Nothing+ }+ )++foldl1Column ::+ forall a.+ (Columnable a) =>+ (a -> a -> a) -> Column -> Either DataFrameException a+foldl1Column f = \case+ BoxedColumn _ column -> foldl1Worker column+ UnboxedColumn _ column -> foldl1Worker column+ c@(PackedText _ _) -> foldl1Column f (materializePacked c)+ c@(MergedColumn _ _) -> foldl1Column f (materializeMerged c)+ where+ foldl1Worker ::+ forall c v.+ (Typeable c, VG.Vector v c) =>+ v c ->+ Either DataFrameException a+ foldl1Worker vec = case testEquality (typeRep @a) (typeRep @c) of+ Just Refl -> pure $ VG.foldl1' f vec+ Nothing ->+ Left $+ TypeMismatchException+ ( MkTypeErrorContext+ { userType = Right (typeRep @a)+ , expectedType = Right (typeRep @c)+ , callingFunctionName = Just "foldl1Column"+ , errorColumnName = Nothing+ }+ )++{- | O(n) Seedless fold over groups using the first element of each group as seed.+Like 'foldDirectGroups' but for the case where no initial accumulator is available.+-}+foldl1DirectGroups ::+ forall a.+ (Columnable a) =>+ (a -> a -> a) ->+ Column ->+ VU.Vector Int ->+ VU.Vector Int ->+ Either DataFrameException Column+foldl1DirectGroups f col valueIndices offsets+ | VU.length offsets <= 1 = pure $ fromVector @a VB.empty+ | otherwise = case col of+ UnboxedColumn _ (vec :: VU.Vector d) -> UnboxedColumn Nothing <$> foldl1Worker vec+ BoxedColumn _ (vec :: VB.Vector d) -> BoxedColumn Nothing <$> foldl1Worker vec+ PackedText _ _ -> foldl1DirectGroups f (materializePacked col) valueIndices offsets+ MergedColumn _ _ -> foldl1DirectGroups f (materializeMerged col) valueIndices offsets+ where+ foldl1Worker ::+ forall c v.+ (Typeable c, VG.Vector v c) =>+ v c ->+ Either DataFrameException (v c)+ foldl1Worker vec = case testEquality (typeRep @a) (typeRep @c) of+ Just Refl ->+ Right $+ VG.generate (VU.length offsets - 1) foldGroup+ where+ foldGroup k =+ let !s = VU.unsafeIndex offsets k+ !e = VU.unsafeIndex offsets (k + 1)+ !seed = VG.unsafeIndex vec (VU.unsafeIndex valueIndices s)+ in go (s + 1) e seed+ go !i !e !acc+ | i >= e = acc+ | otherwise =+ go (i + 1) e $!+ f acc (VG.unsafeIndex vec (VU.unsafeIndex valueIndices i))+ Nothing ->+ Left $+ TypeMismatchException+ MkTypeErrorContext+ { userType = Right (typeRep @a)+ , expectedType = Right (typeRep @c)+ , callingFunctionName = Just "foldl1DirectGroups"+ , errorColumnName = Nothing+ }+{-# INLINEABLE foldl1DirectGroups #-}++{- | O(n) fold over groups by scanning the column linearly (rowToGroup[i] = group+of row i). Random writes hit the small per-group accumulator array; when @acc@ is+unboxable that array is unboxed, avoiding pointer indirection.+-}+foldLinearGroups ::+ forall b acc.+ (Columnable b, Columnable acc) =>+ (acc -> b -> acc) ->+ acc ->+ Column ->+ VU.Vector Int ->+ Int ->+ Either DataFrameException Column+foldLinearGroups f seed col rowToGroup nGroups+ | nGroups == 0 = Right (fromVector @acc VB.empty)+ | otherwise = case col of+ UnboxedColumn _ (vec :: VU.Vector d) -> foldLinearWorker vec+ BoxedColumn _ (vec :: VB.Vector d) -> foldLinearWorker vec+ PackedText _ _ ->+ foldLinearGroups f seed (materializePacked col) rowToGroup nGroups+ MergedColumn _ _ ->+ foldLinearGroups f seed (materializeMerged col) rowToGroup nGroups+ where+ foldLinearWorker ::+ forall c v.+ (Typeable c, VG.Vector v c) =>+ v c ->+ Either DataFrameException Column+ foldLinearWorker vec = case testEquality (typeRep @b) (typeRep @c) of+ Just Refl ->+ Right $+ unsafePerformIO $+ runWith+ ( \readAt writeAt ->+ VG.iforM_ vec $ \row x -> do+ let !k = VG.unsafeIndex rowToGroup row+ cur <- readAt k+ writeAt k $! f cur x+ )+ Nothing ->+ Left $+ TypeMismatchException+ MkTypeErrorContext+ { userType = Right (typeRep @b)+ , expectedType = Right (typeRep @c)+ , callingFunctionName = Just "foldLinearGroups"+ , errorColumnName = Nothing+ }++ runWith :: ((Int -> IO acc) -> (Int -> acc -> IO ()) -> IO ()) -> IO Column+ runWith body = case sUnbox @acc of+ STrue -> do+ accs <- VUM.replicate nGroups seed+ body (VUM.unsafeRead accs) (VUM.unsafeWrite accs)+ UnboxedColumn Nothing <$> VU.unsafeFreeze accs+ SFalse -> do+ accs <- VBM.replicate nGroups seed+ body (VBM.unsafeRead accs) (VBM.unsafeWrite accs)+ fromVector @acc <$> VB.unsafeFreeze accs+ {-# INLINE runWith #-}+{-# INLINEABLE foldLinearGroups #-}++headColumn :: forall a. (Columnable a) => Column -> Either DataFrameException a+headColumn = \case+ BoxedColumn _ col -> headWorker col+ UnboxedColumn _ col -> headWorker col+ c@(PackedText _ _) -> headColumn (materializePacked c)+ c@(MergedColumn _ _) -> headColumn (mergedHead c)+ where+ headWorker ::+ forall c v.+ (Typeable c, VG.Vector v c) =>+ v c ->+ Either DataFrameException a+ headWorker vec = case testEquality (typeRep @a) (typeRep @c) of+ Just Refl ->+ if VG.null vec+ then Left (EmptyDataSetException "headColumn")+ else pure (VG.head vec)+ Nothing ->+ Left $+ TypeMismatchException+ ( MkTypeErrorContext+ { userType = Right (typeRep @a)+ , expectedType = Right (typeRep @c)+ , callingFunctionName = Just "headColumn"+ , errorColumnName = Nothing+ }+ )++-- | An internal, column version of zip.+zipColumns :: Column -> Column -> Column+zipColumns l@(MergedColumn _ _) r = zipColumns (materializeMerged l) r+zipColumns l r@(MergedColumn _ _) = zipColumns l (materializeMerged r)+zipColumns l@(PackedText _ _) r = zipColumns (materializePacked l) r+zipColumns l r@(PackedText _ _) = zipColumns l (materializePacked r)+zipColumns (BoxedColumn _ column) (BoxedColumn _ other) = BoxedColumn Nothing (VG.zip column other)+zipColumns (BoxedColumn _ column) (UnboxedColumn _ other) =+ BoxedColumn+ Nothing+ ( VB.generate+ (min (VG.length column) (VG.length other))+ (\i -> (column VG.! i, other VG.! i))+ )+zipColumns (UnboxedColumn _ column) (BoxedColumn _ other) =+ BoxedColumn+ Nothing+ ( VB.generate+ (min (VG.length column) (VG.length other))+ (\i -> (column VG.! i, other VG.! i))+ )+zipColumns (UnboxedColumn _ column) (UnboxedColumn _ other) = UnboxedColumn Nothing (VG.zip column other)+{-# INLINE zipColumns #-}++-- | An internal, column version of zipWith.+zipWithColumns ::+ forall a b c.+ (Columnable a, Columnable b, Columnable c) =>+ (a -> b -> c) -> Column -> Column -> Either DataFrameException Column+zipWithColumns f (UnboxedColumn bmL (column :: VU.Vector d)) (UnboxedColumn bmR (other :: VU.Vector e)) = case testEquality (typeRep @a) (typeRep @d) of+ Just Refl -> case testEquality (typeRep @b) (typeRep @e) of+ Just Refl+ | isNothing bmL+ , isNothing bmR ->+ pure $ case sUnbox @c of+ STrue ->+ let !n = min (VU.length column) (VU.length other)+ in UnboxedColumn Nothing $+ parGenerateUnboxed n $ \i ->+ f (VU.unsafeIndex column i) (VU.unsafeIndex other i)+ SFalse -> fromVector $ VB.zipWith f (VG.convert column) (VG.convert other)+ _ -> zipWithColumnsGeneral f (UnboxedColumn bmL column) (UnboxedColumn bmR other)+ Nothing -> zipWithColumnsGeneral f (UnboxedColumn bmL column) (UnboxedColumn bmR other)+-- TODO: mchavinda - reuse pattern from interpret where we augment the+-- error at the end.+zipWithColumns f left right = zipWithColumnsGeneral f left right++zipWithColumnsGeneral ::+ forall a b c.+ (Columnable a, Columnable b, Columnable c) =>+ (a -> b -> c) -> Column -> Column -> Either DataFrameException Column+zipWithColumnsGeneral f left right = case toVector @a left of+ Left (TypeMismatchException context) ->+ Left $+ TypeMismatchException (context{callingFunctionName = Just "zipWithColumns"})+ Left e -> Left e+ Right left' -> case toVector @b right of+ Left (TypeMismatchException context) ->+ Left $+ TypeMismatchException (context{callingFunctionName = Just "zipWithColumns"})+ Left e -> Left e+ Right right' -> pure $ fromVector $ VB.zipWith f left' right'+{-# INLINE zipWithColumnsGeneral #-}+{-# INLINE zipWithColumns #-}++-- writeColumn and freezeColumn' (CSV-ingest helpers) moved to+-- DataFrame.IO.Internal.MutableColumn so the core column module does not+-- need to depend on DataFrame.Internal.Parsing.++{- | Freeze a mutable column into an @Either Text a@ column: every recorded+null position becomes @Left rawText@ (preserving the original input), every+other position becomes @Right v@. Used by CSV readers under 'EitherRead' mode.+-}+freezeColumnEither :: [(Int, T.Text)] -> MutableColumn -> IO Column+freezeColumnEither nulls (MBoxedColumn col) = do+ frozen <- VB.unsafeFreeze col+ let nullMap = nulls+ pure $+ BoxedColumn Nothing $+ VB.imap+ ( \i v -> case lookup i nullMap of+ Just t -> Left t+ Nothing -> Right v+ )+ frozen+freezeColumnEither nulls (MUnboxedColumn col) = do+ c <- VU.unsafeFreeze col+ let nullMap = nulls+ pure $+ BoxedColumn Nothing $+ VB.generate (VU.length c) $ \i ->+ case lookup i nullMap of+ Just t -> Left t+ Nothing -> Right (c VU.! i)+{-# INLINE freezeColumnEither #-}++{- | Promote a non-nullable column to a nullable one (add an all-valid bitmap).+No-op when already nullable.+-}+ensureOptional :: Column -> Column+ensureOptional c@(MergedColumn _ _) = ensureOptional (materializeMerged c)+ensureOptional c@(BoxedColumn (Just _) _) = c+ensureOptional (BoxedColumn Nothing col) =+ BoxedColumn (Just (allValidBitmap (VB.length col))) col+ensureOptional c@(UnboxedColumn (Just _) _) = c+ensureOptional (UnboxedColumn Nothing col) =+ UnboxedColumn (Just (allValidBitmap (VU.length col))) col+ensureOptional c@(PackedText (Just _) _) = c+ensureOptional (PackedText Nothing p) =+ PackedText (Just (allValidBitmap (packedLength p))) p++-- | Fills the end of a column, up to n, with null rows. Does nothing if column has length >= n.+expandColumn :: Int -> Column -> Column+expandColumn n c@(MergedColumn a b)+ | n <= min (columnLength a) (columnLength b) = c+ | otherwise = expandColumn n (materializeMerged c)+expandColumn n c@(PackedText _ p)+ | n <= packedLength p = c+ | otherwise = expandColumn n (materializePacked c)+expandColumn n column@(BoxedColumn bm col)+ | n <= VG.length col = column+ | otherwise =+ let extra = n - VG.length col+ newBm = case bm of+ Nothing -> Just (buildBitmapFromNulls' n (VU.enumFromN (VG.length col) extra))+ Just b ->+ Just+ (bitmapConcat (VG.length col) b extra (VU.replicate ((extra + 7) `shiftR` 3) 0))+ newCol = col <> VB.replicate extra (errorWithoutStackTrace "expandColumn: null slot")+ in BoxedColumn newBm newCol+expandColumn n column@(UnboxedColumn bm col)+ | n <= VG.length col = column+ | otherwise =+ let extra = n - VG.length col+ newBm = case bm of+ Nothing -> Just (buildBitmapFromNulls' n (VU.enumFromN (VG.length col) extra))+ Just b ->+ Just+ (bitmapConcat (VG.length col) b extra (VU.replicate ((extra + 7) `shiftR` 3) 0))+ newCol = runST $ do+ mv <- VUM.new n+ VU.imapM_ (VUM.unsafeWrite mv) col+ VU.unsafeFreeze mv+ in UnboxedColumn newBm newCol++-- | Fills the beginning of a column, up to n, with null rows. Does nothing if column has length >= n.+leftExpandColumn :: Int -> Column -> Column+leftExpandColumn n c@(MergedColumn a b)+ | n <= min (columnLength a) (columnLength b) = c+ | otherwise = leftExpandColumn n (materializeMerged c)+leftExpandColumn n c@(PackedText _ p)+ | n <= packedLength p = c+ | otherwise = leftExpandColumn n (materializePacked c)+leftExpandColumn n column@(BoxedColumn bm col)+ | n <= VG.length col = column+ | otherwise =+ let extra = n - VG.length col+ origLen = VG.length col+ newBm = case bm of+ Nothing -> Just (buildBitmapFromNulls' n (VU.enumFromN 0 extra))+ Just b ->+ let nullPart = VU.replicate ((extra + 7) `shiftR` 3) 0+ in Just (bitmapConcat extra nullPart origLen b)+ newCol =+ VB.replicate extra (errorWithoutStackTrace "leftExpandColumn: null slot") <> col+ in BoxedColumn newBm newCol+leftExpandColumn n column@(UnboxedColumn bm col)+ | n <= VG.length col = column+ | otherwise =+ let extra = n - VG.length col+ origLen = VG.length col+ newBm = case bm of+ Nothing -> Just (buildBitmapFromNulls' n (VU.enumFromN 0 extra))+ Just b ->+ let nullPart = VU.replicate ((extra + 7) `shiftR` 3) 0+ in Just (bitmapConcat extra nullPart origLen b)+ newCol = runST $ do+ mv <- VUM.new n+ VU.imapM_ (\i x -> VUM.unsafeWrite mv (extra + i) x) col+ VU.unsafeFreeze mv+ in UnboxedColumn newBm newCol++{- | Concatenates two columns.+Returns Nothing if the columns are of different types.+-}+mappendColumns :: Column -> Column -> Either DataFrameException Column+mappendColumns left right = case (left, right) of+ (MergedColumn _ _, _) -> mappendColumns (materializeMerged left) right+ (_, MergedColumn _ _) -> mappendColumns left (materializeMerged right)+ (PackedText _ _, _) -> mappendColumns (materializePacked left) right+ (_, PackedText _ _) -> mappendColumns left (materializePacked right)+ (BoxedColumn bmL l, BoxedColumn bmR r) -> case testEquality (typeOf l) (typeOf r) of+ Just Refl ->+ let newBm = case (bmL, bmR) of+ (Nothing, Nothing) -> Nothing+ (Just bl, Nothing) ->+ Just+ (bitmapConcat (VB.length l) bl (VB.length r) (allValidBitmap (VB.length r)))+ (Nothing, Just br) ->+ Just+ (bitmapConcat (VB.length l) (allValidBitmap (VB.length l)) (VB.length r) br)+ (Just bl, Just br) -> Just (bitmapConcat (VB.length l) bl (VB.length r) br)+ in pure (BoxedColumn newBm (l <> r))+ Nothing -> Left (mismatchErr (typeOf r) (typeOf l))+ (UnboxedColumn bmL l, UnboxedColumn bmR r) -> case testEquality (typeOf l) (typeOf r) of+ Just Refl ->+ let newBm = case (bmL, bmR) of+ (Nothing, Nothing) -> Nothing+ (Just bl, Nothing) ->+ Just+ (bitmapConcat (VU.length l) bl (VU.length r) (allValidBitmap (VU.length r)))+ (Nothing, Just br) ->+ Just+ (bitmapConcat (VU.length l) (allValidBitmap (VU.length l)) (VU.length r) br)+ (Just bl, Just br) -> Just (bitmapConcat (VU.length l) bl (VU.length r) br)+ in pure (UnboxedColumn newBm (l <> r))+ Nothing -> Left (mismatchErr (typeOf r) (typeOf l))+ _ -> Left (mismatchErr (typeOf right) (typeOf left))+ where+ mismatchErr ::+ forall (x :: Type) (y :: Type). TypeRep x -> TypeRep y -> DataFrameException+ mismatchErr ta tb =+ withTypeable ta $+ withTypeable tb $+ TypeMismatchException+ ( MkTypeErrorContext+ { userType = Right ta+ , expectedType = Right tb+ , callingFunctionName = Just "mappendColumns"+ , errorColumnName = Nothing+ }+ )++{- | Like 'mappendColumns' but also combines columns of different types by wrapping+values in 'Either' (e.g. @[1,2]@ and @["a","b"]@ become+@[Left 1, Left 2, Right "a", Right "b"]@).+-}++{- | O(n) Concatenate a list of same-type columns in a single allocation.+All columns must have the same constructor and element type (as they will+within a single Parquet column). Calls 'error' on mismatch.+-}+concatManyColumns :: [Column] -> Column+concatManyColumns [] = fromList ([] :: [Maybe Int])+concatManyColumns [c] = c+concatManyColumns all'+ | any isMergedColumn all' =+ concatManyColumns (map materializeMerged all')+ | any isPackedText all' =+ concatManyColumns (map materializePacked all')+concatManyColumns (c0 : cs) = case c0 of+ BoxedColumn bm0 v0 ->+ let getCol (BoxedColumn bm v) = case testEquality (typeOf v0) (typeOf v) of+ Just Refl -> (bm, v)+ Nothing -> error "concatManyColumns: BoxedColumn type mismatch"+ getCol _ = error "concatManyColumns: column constructor mismatch"+ rest = map getCol cs+ allVecs = v0 : map snd rest+ allBms = bm0 : map fst rest+ newBm+ | all isNothing allBms = Nothing+ | otherwise =+ let pairs = zip allVecs allBms+ expandedBms = map (\(v, mb) -> fromMaybe (allValidBitmap (VB.length v)) mb) pairs+ go b1 n1 b2 n2 = bitmapConcat n1 b1 n2 b2+ concatBms [] = VU.empty+ concatBms [(b, _v)] = b+ concatBms ((b1, v1) : (b2, v2) : rest') =+ let merged = go b1 (VB.length v1) b2 (VB.length v2)+ in concatBms ((merged, v1 <> v2) : rest')+ in Just $ concatBms (zip expandedBms allVecs)+ in BoxedColumn newBm (VB.concat allVecs)+ UnboxedColumn bm0 v0 ->+ let getCol (UnboxedColumn bm v) = case testEquality (typeOf v0) (typeOf v) of+ Just Refl -> (bm, v)+ Nothing -> error "concatManyColumns: UnboxedColumn type mismatch"+ getCol _ = error "concatManyColumns: column constructor mismatch"+ rest = map getCol cs+ allVecs = v0 : map snd rest+ allBms = bm0 : map fst rest+ newBm+ | all isNothing allBms = Nothing+ | otherwise =+ let pairs = zip allVecs allBms+ expandedBms = map (\(v, mb) -> fromMaybe (allValidBitmap (VU.length v)) mb) pairs+ go b1 n1 b2 n2 = bitmapConcat n1 b1 n2 b2+ concatBms [] = VU.empty+ concatBms [(b, _)] = b+ concatBms ((b1, v1) : (b2, v2) : rest') =+ let merged = go b1 (VU.length v1) b2 (VU.length v2)+ in concatBms ((merged, v1 <> v2) : rest')+ in Just $ concatBms (zip expandedBms allVecs)+ in UnboxedColumn newBm (VU.concat allVecs)+ PackedText _ _ -> concatManyColumns (map materializePacked (c0 : cs))+ MergedColumn _ _ -> concatManyColumns (map materializeMerged (c0 : cs))++mappendColumnsEither :: Column -> Column -> Column+mappendColumnsEither l@(MergedColumn _ _) r =+ mappendColumnsEither (materializeMerged l) r+mappendColumnsEither l r@(MergedColumn _ _) =+ mappendColumnsEither l (materializeMerged r)+mappendColumnsEither l@(PackedText _ _) r = mappendColumnsEither (materializePacked l) r+mappendColumnsEither l r@(PackedText _ _) = mappendColumnsEither l (materializePacked r)+mappendColumnsEither (BoxedColumn bmL left) (BoxedColumn bmR right) = case testEquality (typeOf left) (typeOf right) of+ Nothing ->+ BoxedColumn Nothing $ fmap Left left <> fmap Right right+ Just Refl ->+ let newBm = case (bmL, bmR) of+ (Nothing, Nothing) -> Nothing+ (Just bl, Nothing) ->+ Just+ ( bitmapConcat+ (VB.length left)+ bl+ (VB.length right)+ (allValidBitmap (VB.length right))+ )+ (Nothing, Just br) ->+ Just+ ( bitmapConcat+ (VB.length left)+ (allValidBitmap (VB.length left))+ (VB.length right)+ br+ )+ (Just bl, Just br) -> Just (bitmapConcat (VB.length left) bl (VB.length right) br)+ in BoxedColumn newBm $ left <> right+mappendColumnsEither (UnboxedColumn bmL left) (UnboxedColumn bmR right) = case testEquality (typeOf left) (typeOf right) of+ Nothing ->+ BoxedColumn Nothing $+ fmap Left (VG.convert left) <> fmap Right (VG.convert right)+ Just Refl ->+ let newBm = case (bmL, bmR) of+ (Nothing, Nothing) -> Nothing+ (Just bl, Nothing) ->+ Just+ ( bitmapConcat+ (VU.length left)+ bl+ (VU.length right)+ (allValidBitmap (VU.length right))+ )+ (Nothing, Just br) ->+ Just+ ( bitmapConcat+ (VU.length left)+ (allValidBitmap (VU.length left))+ (VU.length right)+ br+ )+ (Just bl, Just br) -> Just (bitmapConcat (VU.length left) bl (VU.length right) br)+ in UnboxedColumn newBm $ left <> right+mappendColumnsEither (BoxedColumn _ left) (UnboxedColumn _ right) =+ BoxedColumn Nothing $ fmap Left left <> fmap Right (VG.convert right)+mappendColumnsEither (UnboxedColumn _ left) (BoxedColumn _ right) =+ BoxedColumn Nothing $ fmap Left (VG.convert left) <> fmap Right right++-------------------------------------------------------------------------------+-- Fused multi-column gather+-------------------------------------------------------------------------------++{- | Gather ONE in-bounds index vector through several columns in a single+parallel pass. Result columns are identical to @map (atIndicesStable ixs)@;+columns whose shape the fused kernel does not cover (bitmapped, boxed, merged,+unusual element types) fall back to per-column 'atIndicesStable'. All fused+outputs are backed by one shared deferred computation: forcing ANY of them+runs the single pass that fills ALL of them. The pass reads the index vector+once per block instead of once per column, and every iteration of a block+issues each column's independent random load back-to-back, so their cache/TLB+misses overlap instead of forming one latency chain per column (the aggregate+key materialization of a 1e8-group result was 6 sequential latency-bound+passes without this).+-}+atIndicesStableMulti :: VU.Vector Int -> [Column] -> [Column]+atIndicesStableMulti ixs cols =+ let specs = map mgClassify cols+ nFused = length [() | Just _ <- specs]+ in if nFused < 2+ then map (atIndicesStable ixs) cols+ else+ let fused = multiGatherRun ixs (catMaybes specs)+ go [] _ = []+ go (Nothing : ss) !k = atIndicesStable ixs (cols !! k) : go ss (k + 1)+ go (Just _ : ss) !k =+ let !r = mgRank k+ in (fused VB.! r) : go ss (k + 1)+ -- fused-output rank of column position k.+ mgRank k = length [() | Just _ <- take k specs]+ in go specs 0++-- | One fusable source column shape (all bitmap-free).+data MGSpec+ = -- | Clean unboxed 'Int' payload.+ MGInt !(VU.Vector Int)+ | -- | Clean unboxed 'Double' payload.+ MGDouble !(VU.Vector Double)+ | {- | Packed text with an 'Int32' selector (base row count necessarily+ fits 'Int32'); carries the payload for rebuilding and the canon flag.+ -}+ MGSel32 !PackedTextData !(VU.Vector Int32)+ | -- | Packed text with an 'Int' selector and an 'Int32'-sized base.+ MGSel64To32 !PackedTextData !(VU.Vector Int)+ | -- | Packed text with an 'Int' selector and a wide base.+ MGSel64To64 !PackedTextData !(VU.Vector Int)++mgClassify :: Column -> Maybe MGSpec+mgClassify (UnboxedColumn Nothing (v :: VU.Vector a)) =+ case testEquality (typeRep @a) (typeRep @Int) of+ Just Refl -> Just (MGInt v)+ Nothing -> case testEquality (typeRep @a) (typeRep @Double) of+ Just Refl -> Just (MGDouble v)+ Nothing -> Nothing+mgClassify (PackedText Nothing p) = case ptSel p of+ Just (Sel32 s) -> Just (MGSel32 p s)+ Just (Sel64 s)+ | offCount (ptOffsets p) - 1 <= mgInt32Max -> Just (MGSel64To32 p s)+ | otherwise -> Just (MGSel64To64 p s)+ Nothing -> Nothing+mgClassify _ = Nothing++mgInt32Max :: Int+mgInt32Max = fromIntegral (maxBound :: Int32)++{- | Run the fused pass over the fusable specs; element @r@ of the result is+spec @r@'s gathered column. Deferred as one shared thunk (see+'atIndicesStableMulti'). Pure w.r.t. its immutable inputs, so the+'unsafePerformIO' is safe.+-}+multiGatherRun :: VU.Vector Int -> [MGSpec] -> VB.Vector Column+multiGatherRun ixs specs = unsafePerformIO $ do+ let !n = VU.length ixs+ opened <- mapM (mgOpen n ixs) specs+ let fills = map fst opened+ !block = 4096+ worker !lo !hi+ | lo >= hi = pure ()+ | otherwise = do+ let !e = min hi (lo + block)+ mapM_ (\fill -> fill lo e) fills+ worker e hi+ parallelChunks_ parThreshold n worker+ VB.fromList <$> mapM snd opened+{-# NOINLINE multiGatherRun #-}++{- | Allocate a spec's destination; return its block-fill action and its+finalizer. Selector gathers reproduce 'packedGather' exactly (composition+clamps against the source selector length and base row count); unboxed gathers+reproduce the unclamped 'parBackpermuteUnboxed'.+-}+mgOpen :: Int -> VU.Vector Int -> MGSpec -> IO (Int -> Int -> IO (), IO Column)+mgOpen n ixs spec = case spec of+ MGInt v -> do+ mv <- VUM.unsafeNew n+ pure+ ( mgFillInt ixs v mv+ , UnboxedColumn Nothing <$> VU.unsafeFreeze mv+ )+ MGDouble v -> do+ mv <- VUM.unsafeNew n+ pure+ ( mgFillDouble ixs v mv+ , UnboxedColumn Nothing <$> VU.unsafeFreeze mv+ )+ MGSel32 p s -> do+ mv <- VUM.unsafeNew n+ pure+ ( mgFillSel32 ixs s (offCount (ptOffsets p) - 1) mv+ , (\out -> PackedText Nothing p{ptSel = Just (Sel32 out)}) <$> VU.unsafeFreeze mv+ )+ MGSel64To32 p s -> do+ mv <- VUM.unsafeNew n+ pure+ ( mgFillSel64To32 ixs s (offCount (ptOffsets p) - 1) mv+ , (\out -> PackedText Nothing p{ptSel = Just (Sel32 out)}) <$> VU.unsafeFreeze mv+ )+ MGSel64To64 p s -> do+ mv <- VUM.unsafeNew n+ pure+ ( mgFillSel64To64 ixs s (offCount (ptOffsets p) - 1) mv+ , (\out -> PackedText Nothing p{ptSel = Just (Sel64 out)}) <$> VU.unsafeFreeze mv+ )++mgFillInt ::+ VU.Vector Int -> VU.Vector Int -> VUM.IOVector Int -> Int -> Int -> IO ()+mgFillInt ixs v dst lo hi = go lo+ where+ go !i+ | i >= hi = pure ()+ | otherwise = do+ VUM.unsafeWrite dst i (VU.unsafeIndex v (VU.unsafeIndex ixs i))+ go (i + 1)+{-# NOINLINE mgFillInt #-}++mgFillDouble ::+ VU.Vector Int -> VU.Vector Double -> VUM.IOVector Double -> Int -> Int -> IO ()+mgFillDouble ixs v dst lo hi = go lo+ where+ go !i+ | i >= hi = pure ()+ | otherwise = do+ VUM.unsafeWrite dst i (VU.unsafeIndex v (VU.unsafeIndex ixs i))+ go (i + 1)+{-# NOINLINE mgFillDouble #-}++mgFillSel32 ::+ VU.Vector Int ->+ VU.Vector Int32 ->+ Int ->+ VUM.IOVector Int32 ->+ Int ->+ Int ->+ IO ()+mgFillSel32 ixs s !base dst lo hi = go lo+ where+ !sn = VU.length s+ go !i+ | i >= hi = pure ()+ | otherwise = do+ let !j = VU.unsafeIndex ixs i+ !out =+ if j >= 0 && j < sn+ then+ let !r = fromIntegral (VU.unsafeIndex s j) :: Int+ in if r >= 0 && r < base then fromIntegral r else -1+ else -1+ VUM.unsafeWrite dst i out+ go (i + 1)+{-# NOINLINE mgFillSel32 #-}++mgFillSel64To32 ::+ VU.Vector Int ->+ VU.Vector Int ->+ Int ->+ VUM.IOVector Int32 ->+ Int ->+ Int ->+ IO ()+mgFillSel64To32 ixs s !base dst lo hi = go lo+ where+ !sn = VU.length s+ go !i+ | i >= hi = pure ()+ | otherwise = do+ let !j = VU.unsafeIndex ixs i+ !out =+ if j >= 0 && j < sn+ then+ let !r = VU.unsafeIndex s j+ in if r >= 0 && r < base then fromIntegral r else -1+ else -1+ VUM.unsafeWrite dst i out+ go (i + 1)+{-# NOINLINE mgFillSel64To32 #-}++mgFillSel64To64 ::+ VU.Vector Int -> VU.Vector Int -> Int -> VUM.IOVector Int -> Int -> Int -> IO ()+mgFillSel64To64 ixs s !base dst lo hi = go lo+ where+ !sn = VU.length s+ go !i+ | i >= hi = pure ()+ | otherwise = do+ let !j = VU.unsafeIndex ixs i+ !out =+ if j >= 0 && j < sn+ then+ let !r = VU.unsafeIndex s j+ in if r >= 0 && r < base then r else -1+ else -1+ VUM.unsafeWrite dst i out+ go (i + 1)+{-# NOINLINE mgFillSel64To64 #-}
+ src-internal/DataFrame/Internal/Column/Properties.hs view
@@ -0,0 +1,153 @@+{-# LANGUAGE AllowAmbiguousTypes #-}+{-# LANGUAGE ConstraintKinds #-}+{-# LANGUAGE DataKinds #-}+{-# LANGUAGE FlexibleContexts #-}+{-# LANGUAGE FlexibleInstances #-}+{-# LANGUAGE GADTs #-}+{-# LANGUAGE LambdaCase #-}+{-# LANGUAGE MultiParamTypeClasses #-}+{-# LANGUAGE OverloadedStrings #-}+{-# LANGUAGE PatternSynonyms #-}+{-# LANGUAGE PolyKinds #-}+{-# LANGUAGE RankNTypes #-}+{-# LANGUAGE ScopedTypeVariables #-}+{-# LANGUAGE TypeApplications #-}+{-# LANGUAGE TypeFamilies #-}+{-# LANGUAGE UndecidableInstances #-}++{- |+Predicates and introspection over a 'Column': representation tests, null/type+queries, and the human-readable type descriptions used in error messages.+-}+module DataFrame.Internal.Column.Properties where++import qualified Data.Text as T+import qualified Data.Vector as VB+import qualified Data.Vector.Unboxed as VU++import Data.Kind (Type)+import Data.Maybe (isJust)+import Data.Type.Equality (TestEquality (..))+import DataFrame.Internal.Column.Base+import DataFrame.Internal.Column.Bitmap+import DataFrame.Internal.Column.Types+import DataFrame.Internal.Data.PackedText (packedLength)+import Type.Reflection (+ TypeRep,+ Typeable,+ eqTypeRep,+ typeRep,+ pattern App,+ type (:~:) (Refl),+ type (:~~:) (HRefl),+ )++-- | Whether a column is a 'PackedText'.+isPackedText :: Column -> Bool+isPackedText (PackedText _ _) = True+isPackedText _ = False+{-# INLINE isPackedText #-}++-- | Whether a column is a 'MergedColumn'.+isMergedColumn :: Column -> Bool+isMergedColumn (MergedColumn _ _) = True+isMergedColumn _ = False+{-# INLINE isMergedColumn #-}++-- | Checks if a column contains missing values (has a bitmap).+hasMissing :: Column -> Bool+hasMissing (BoxedColumn (Just _) _) = True+hasMissing (UnboxedColumn (Just _) _) = True+hasMissing (PackedText (Just _) _) = True+hasMissing _ = False++-- | Checks if a column contains only missing values.+allMissing :: Column -> Bool+allMissing (BoxedColumn (Just bm) col) =+ not (VB.null col) && popCountUpTo (VB.length col) bm == 0+allMissing (UnboxedColumn (Just bm) col) =+ not (VU.null col) && popCountUpTo (VU.length col) bm == 0+allMissing (PackedText (Just bm) p) =+ packedLength p > 0 && popCountUpTo (packedLength p) bm == 0+allMissing _ = False++-- | Checks if a column contains numeric values.+isNumeric :: Column -> Bool+isNumeric c@(MergedColumn _ _) = isNumeric (mergedHead c)+isNumeric (UnboxedColumn _ (_vec :: VU.Vector a)) = case sNumeric @a of+ STrue -> True+ _ -> False+isNumeric (BoxedColumn _ (_vec :: VB.Vector a)) = case testEquality (typeRep @a) (typeRep @Integer) of+ Nothing -> False+ Just Refl -> True+isNumeric (PackedText _ _) = False++{- | Whether the column stores element type @a@. For nullable columns, also+'True' when @a = Maybe b@ and the column stores @b@ internally.+-}+hasElemType :: forall a. (Columnable a) => Column -> Bool+hasElemType = \case+ BoxedColumn bm (_column :: VB.Vector b) -> checkBoxed bm (typeRep @b)+ UnboxedColumn bm (_column :: VU.Vector b) -> checkUnboxed bm (typeRep @b)+ PackedText bm _ -> checkBoxed bm (typeRep @T.Text)+ c@(MergedColumn _ _) -> hasElemType @a (mergedHead c)+ where+ directMatch :: forall (b :: Type). TypeRep b -> Bool+ directMatch = isJust . testEquality (typeRep @a)+ checkMaybe :: forall (b :: Type). TypeRep b -> Bool+ checkMaybe tb = case typeRep @a of+ App tMaybe tInner -> case eqTypeRep tMaybe (typeRep @Maybe) of+ Just HRefl -> isJust (testEquality tInner tb)+ Nothing -> False+ _ -> False+ checkBoxed :: forall (b :: Type). Maybe Bitmap -> TypeRep b -> Bool+ checkBoxed bm tb = directMatch tb || (isJust bm && checkMaybe tb)+ checkUnboxed :: forall (b :: Type). Maybe Bitmap -> TypeRep b -> Bool+ checkUnboxed bm tb = directMatch tb || (isJust bm && checkMaybe tb)++-- | An internal/debugging function to get the column type of a column.+columnVersionString :: Column -> String+columnVersionString column = case column of+ BoxedColumn Nothing _ -> "Boxed"+ BoxedColumn (Just _) _ -> "NullableBoxed"+ UnboxedColumn Nothing _ -> "Unboxed"+ UnboxedColumn (Just _) _ -> "NullableUnboxed"+ PackedText Nothing _ -> "Boxed"+ PackedText (Just _) _ -> "NullableBoxed"+ MergedColumn _ _ -> columnVersionString (mergedHead column)++{- | An internal/debugging function to get the type stored in the outermost vector+of a column.+-}+columnTypeString :: Column -> String+columnTypeString column = case column of+ BoxedColumn Nothing (_ :: VB.Vector a) -> show (typeRep @a)+ BoxedColumn (Just _) (_ :: VB.Vector a) -> showMaybeType @a+ UnboxedColumn Nothing (_ :: VU.Vector a) -> show (typeRep @a)+ UnboxedColumn (Just _) (_ :: VU.Vector a) -> showMaybeType @a+ PackedText Nothing _ -> show (typeRep @T.Text)+ PackedText (Just _) _ -> showMaybeType @T.Text+ MergedColumn _ _ -> columnTypeString (mergedHead column)+ where+ showMaybeType :: forall a. (Typeable a) => String+ showMaybeType =+ let s = show (typeRep @a)+ in "Maybe " ++ if ' ' `elem` s then "(" ++ s ++ ")" else s++-- | Whether row @i@ is null, respecting the bitmap.+columnElemIsNull :: Column -> Int -> Bool+columnElemIsNull (BoxedColumn (Just bm) _) i = not (bitmapTestBit bm i)+columnElemIsNull (UnboxedColumn (Just bm) _) i = not (bitmapTestBit bm i)+columnElemIsNull (PackedText (Just bm) _) i = not (bitmapTestBit bm i)+columnElemIsNull _ _ = False++-- | O(n) Gets the number of non-null elements in the column.+numElements :: Column -> Int+numElements (MergedColumn a b) = min (columnLength a) (columnLength b)+numElements (BoxedColumn Nothing xs) = VB.length xs+numElements (BoxedColumn (Just bm) xs) = popCountUpTo (VB.length xs) bm+numElements (UnboxedColumn Nothing xs) = VU.length xs+numElements (UnboxedColumn (Just bm) xs) = popCountUpTo (VU.length xs) bm+numElements (PackedText Nothing p) = packedLength p+numElements (PackedText (Just bm) p) = popCountUpTo (packedLength p) bm+{-# INLINE numElements #-}
+ src-internal/DataFrame/Internal/Column/Types.hs view
@@ -0,0 +1,161 @@+{-# LANGUAGE AllowAmbiguousTypes #-}+{-# LANGUAGE ConstraintKinds #-}+{-# LANGUAGE DataKinds #-}+{-# LANGUAGE DeriveTraversable #-}+{-# LANGUAGE ExistentialQuantification #-}+{-# LANGUAGE FlexibleContexts #-}+{-# LANGUAGE GADTs #-}+{-# LANGUAGE PolyKinds #-}+{-# LANGUAGE RankNTypes #-}+{-# LANGUAGE ScopedTypeVariables #-}+{-# LANGUAGE TypeApplications #-}+{-# LANGUAGE TypeFamilies #-}+{-# LANGUAGE UndecidableInstances #-}++module DataFrame.Internal.Column.Types where++import Data.Int (Int16, Int32, Int64, Int8)+import Data.Kind (Constraint, Type)+import Data.Typeable (Typeable)+import qualified Data.Vector.Unboxed as VU+import Data.Word (Word16, Word32, Word64, Word8)++type Columnable' a = (Typeable a, Show a, Eq a)++{- | Inline replacement for @Data.These.These@ to keep @dataframe-core@ free+of the @these@ package dependency. Only the three constructors and the+derived classes are used internally.+-}+data These a b = This a | That b | These a b+ deriving (Eq, Ord, Show, Read, Functor, Foldable, Traversable)++{- | A type with column representations used to select the+"right" representation when specializing the `toColumn` function.+-}+data Rep+ = RBoxed+ | RUnboxed+ | RNullableBoxed++-- | Type-level if statement.+type family If (cond :: Bool) (yes :: k) (no :: k) :: k where+ If 'True yes _ = yes+ If 'False _ no = no++-- | All unboxable types (according to the `vector` package).+type family Unboxable (a :: Type) :: Bool where+ Unboxable Int = 'True+ Unboxable Int8 = 'True+ Unboxable Int16 = 'True+ Unboxable Int32 = 'True+ Unboxable Int64 = 'True+ Unboxable Word = 'True+ Unboxable Word8 = 'True+ Unboxable Word16 = 'True+ Unboxable Word32 = 'True+ Unboxable Word64 = 'True+ Unboxable Char = 'True+ Unboxable Bool = 'True+ Unboxable Double = 'True+ Unboxable Float = 'True+ Unboxable _ = 'False++type family Numeric (a :: Type) :: Bool where+ Numeric Integer = 'True+ Numeric Int = 'True+ Numeric Int8 = 'True+ Numeric Int16 = 'True+ Numeric Int32 = 'True+ Numeric Int64 = 'True+ Numeric Word = 'True+ Numeric Word8 = 'True+ Numeric Word16 = 'True+ Numeric Word32 = 'True+ Numeric Word64 = 'True+ Numeric Double = 'True+ Numeric Float = 'True+ Numeric _ = 'False++-- | Compute the column representation tag for any 'a'.+type family KindOf a :: Rep where+ KindOf (Maybe a) = 'RNullableBoxed+ KindOf a = If (Unboxable a) 'RUnboxed 'RBoxed++-- | Type-level boolean for constraint/type comparison.+data SBool (b :: Bool) where+ STrue :: SBool 'True+ SFalse :: SBool 'False++-- | The runtime witness for our type-level branching.+class SBoolI (b :: Bool) where+ sbool :: SBool b++instance SBoolI 'True where sbool = STrue+instance SBoolI 'False where sbool = SFalse++-- | Runtime witness for whether @a@ is unboxable.+sUnbox :: forall a. (SBoolI (Unboxable a)) => SBool (Unboxable a)+sUnbox = sbool @(Unboxable a)++sNumeric :: forall a. (SBoolI (Numeric a)) => SBool (Numeric a)+sNumeric = sbool @(Numeric a)++type family When (flag :: Bool) (c :: Constraint) :: Constraint where+ When 'True c = c+ When 'False c = ()++type UnboxIf a = When (Unboxable a) (VU.Unbox a)++type family IntegralTypes (a :: Type) :: Bool where+ IntegralTypes Integer = 'True+ IntegralTypes Int = 'True+ IntegralTypes Int8 = 'True+ IntegralTypes Int16 = 'True+ IntegralTypes Int32 = 'True+ IntegralTypes Int64 = 'True+ IntegralTypes Word = 'True+ IntegralTypes Word8 = 'True+ IntegralTypes Word16 = 'True+ IntegralTypes Word32 = 'True+ IntegralTypes Word64 = 'True+ IntegralTypes _ = 'False++sIntegral :: forall a. (SBoolI (IntegralTypes a)) => SBool (IntegralTypes a)+sIntegral = sbool @(IntegralTypes a)++type IntegralIf a = When (IntegralTypes a) (Integral a)++type family FloatingTypes (a :: Type) :: Bool where+ FloatingTypes Float = 'True+ FloatingTypes Double = 'True+ FloatingTypes _ = 'False++sFloating :: forall a. (SBoolI (FloatingTypes a)) => SBool (FloatingTypes a)+sFloating = sbool @(FloatingTypes a)++type FloatingIf a = When (FloatingTypes a) (Real a, Fractional a)++{- | Numeric type promotion: resolves the common type for mixed arithmetic.+Double dominates over Float/Int; Float dominates over Int; same types stay unchanged.+-}+type family Promote (a :: Type) (b :: Type) :: Type where+ Promote a a = a+ Promote Double _ = Double+ Promote _ Double = Double+ Promote Float _ = Float+ Promote _ Float = Float+ Promote Int64 _ = Int64+ Promote _ Int64 = Int64+ Promote Int32 _ = Int32+ Promote _ Int32 = Int32+ Promote a _ = a++{- | Like 'Promote', but integral × integral → Double for use with './' .+Double\/Float still dominate; any two integral types (same or mixed) become Double.+-}+type family PromoteDiv (a :: Type) (b :: Type) :: Type where+ PromoteDiv Double _ = Double+ PromoteDiv _ Double = Double+ PromoteDiv Float _ = Float+ PromoteDiv _ Float = Float+ PromoteDiv _ _ = Double
− src-internal/DataFrame/Internal/ColumnBuilder.hs
@@ -1,299 +0,0 @@-{-# LANGUAGE BangPatterns #-}-{-# LANGUAGE FlexibleContexts #-}-{-# LANGUAGE FlexibleInstances #-}-{-# LANGUAGE UndecidableInstances #-}--{- | Mutable, growable column builders for high-throughput ingest. No-per-append @IORef@ traffic: hot counters live in an unboxed vector, payloads-double on demand, and validity is only materialized once a null is seen.--}-module DataFrame.Internal.ColumnBuilder (- ColumnBuilder (..),- NumBuilder,- IntBuilder,- DoubleBuilder,- TextBuilder,- TextChunk (..),- newIntBuilder,- newDoubleBuilder,- newNumBuilder,- newTextBuilder,- appendInt,- appendDouble,- appendNum,- appendText,- appendTextSlice,- appendTextSliceFromPtr,- freezeTextChunk,- mergeColumns,- mergeTextChunks,-) where--import qualified Data.Text as T-import qualified Data.Text.Array as A-import qualified Data.Vector.Unboxed as VU-import qualified Data.Vector.Unboxed.Mutable as VUM--import Control.Monad (when)-import Control.Monad.ST (ST)-import Data.Bits (shiftR)-import Data.STRef-import Data.Text.Internal (Text (..))-import Data.Word (Word8)-import DataFrame.Internal.Column hiding (mergeColumns)-import DataFrame.Internal.ColumnMerge (- TextChunk (..),- mergeColumns,- mergeTextChunks,- packValidity,- )-import Foreign.Ptr (Ptr)--{- | Operations shared by all column builders. A builder must not be used-again after 'freezeBuilder' (its storage is frozen in place, not copied).--}-class ColumnBuilder b where- -- | Append a null row (sentinel payload + invalid bit).- appendNull :: b s -> ST s ()-- -- | Rows appended so far.- builderLength :: b s -> ST s Int-- -- | Freeze into a fully-forced 'Column'; bitmap only when a null was seen.- freezeBuilder :: b s -> ST s Column---- Counter slots shared by the builders: rows, any-null flag, text bytes used.-cRows, cAnyNull, cBytes :: Int-cRows = 0-cAnyNull = 1-cBytes = 2--{- | Builder for unboxed numeric payloads ('Int', 'Double', ...). 'nbNull'-is the sentinel written into null slots (protected by the bitmap).--}-data NumBuilder a s = NumBuilder- { nbNull :: !a- , nbCounters :: !(VUM.MVector s Int)- , nbArrays :: !(STRef s (NumArrays a s))- }--data NumArrays a s = NumArrays- { naData :: !(VUM.MVector s a)- , naValid :: !(VUM.MVector s Word8)- }--type IntBuilder = NumBuilder Int--type DoubleBuilder = NumBuilder Double---- | New numeric builder with a row-capacity hint and a null sentinel.-newNumBuilder :: (VU.Unbox a) => a -> Int -> ST s (NumBuilder a s)-newNumBuilder nullValue hint = do- let cap = max 16 hint- counters <- VUM.replicate 2 0- dat <- VUM.unsafeNew cap- val <- VUM.unsafeNew cap- NumBuilder nullValue counters <$> newSTRef (NumArrays dat val)--newIntBuilder :: Int -> ST s (IntBuilder s)-newIntBuilder = newNumBuilder 0--newDoubleBuilder :: Int -> ST s (DoubleBuilder s)-newDoubleBuilder = newNumBuilder 0--appendNum :: (VU.Unbox a) => NumBuilder a s -> a -> ST s ()-appendNum b !x = do- n <- VUM.unsafeRead (nbCounters b) cRows- anyNull <- VUM.unsafeRead (nbCounters b) cAnyNull- NumArrays dat val <- reserveNum b n- VUM.unsafeWrite dat n x- when (anyNull /= 0) $ VUM.unsafeWrite val n 1- VUM.unsafeWrite (nbCounters b) cRows (n + 1)-{-# INLINE appendNum #-}--appendInt :: IntBuilder s -> Int -> ST s ()-appendInt = appendNum-{-# INLINE appendInt #-}--appendDouble :: DoubleBuilder s -> Double -> ST s ()-appendDouble = appendNum-{-# INLINE appendDouble #-}---- Fetch the arrays, growing (doubling) first if row @n@ would not fit.-reserveNum :: (VU.Unbox a) => NumBuilder a s -> Int -> ST s (NumArrays a s)-reserveNum b n = do- arrs <- readSTRef (nbArrays b)- if n < VUM.length (naData arrs) then pure arrs else growNum b arrs-{-# INLINE reserveNum #-}--growNum ::- (VU.Unbox a) => NumBuilder a s -> NumArrays a s -> ST s (NumArrays a s)-growNum b (NumArrays dat val) = do- let cap = VUM.length dat- dat' <- VUM.unsafeGrow dat cap- val' <- VUM.unsafeGrow val cap- let arrs = NumArrays dat' val'- writeSTRef (nbArrays b) arrs- pure arrs--instance (Columnable a, VU.Unbox a) => ColumnBuilder (NumBuilder a) where- appendNull b = do- n <- VUM.unsafeRead (nbCounters b) cRows- anyNull <- VUM.unsafeRead (nbCounters b) cAnyNull- NumArrays dat val <- reserveNum b n- VUM.unsafeWrite dat n (nbNull b)- when (anyNull == 0) $ do- VUM.set (VUM.slice 0 n val) 1- VUM.unsafeWrite (nbCounters b) cAnyNull 1- VUM.unsafeWrite val n 0- VUM.unsafeWrite (nbCounters b) cRows (n + 1)- {-# INLINE appendNull #-}-- builderLength b = VUM.unsafeRead (nbCounters b) cRows-- freezeBuilder b = do- n <- VUM.unsafeRead (nbCounters b) cRows- anyNull <- VUM.unsafeRead (nbCounters b) cAnyNull- NumArrays dat val <- readSTRef (nbArrays b)- !vs <- freezeTrimmed n dat- if anyNull /= 0- then do- !bm <- packValidity n val- pure $! UnboxedColumn (Just bm) vs- else pure $! UnboxedColumn Nothing vs---- Zero-copy freeze; copies to exact size when slack exceeds a quarter of n.-freezeTrimmed :: (VU.Unbox a) => Int -> VUM.MVector s a -> ST s (VU.Vector a)-freezeTrimmed n mv- | VUM.length mv - n <= n `shiftR` 2 = VU.unsafeFreeze (VUM.slice 0 n mv)- | otherwise = VU.freeze (VUM.slice 0 n mv)--{- | Builder for 'Text' columns. All field bytes go into one exponentially-grown byte array; rows are recorded as offsets, so an append is a memcpy-and freezing slices 'Text' values off the shared array without copying.--}-data TextBuilder s = TextBuilder- { tbCounters :: !(VUM.MVector s Int)- , tbArrays :: !(STRef s (TextArrays s))- }--data TextArrays s = TextArrays- { taBytes :: !(A.MArray s)- , taByteCap :: !Int- , taOffsets :: !(VUM.MVector s Int)- -- ^ Row @i@ spans bytes @[offsets!i, offsets!(i+1))@.- , taValid :: !(VUM.MVector s Word8)- }---- | New text builder with row-count and total-byte capacity hints.-newTextBuilder :: Int -> Int -> ST s (TextBuilder s)-newTextBuilder rowHint byteHint = do- let rcap = max 16 rowHint- bcap = max 64 byteHint- counters <- VUM.replicate 3 0- bytes <- A.new bcap- offsets <- VUM.unsafeNew (rcap + 1)- VUM.unsafeWrite offsets 0 0- val <- VUM.unsafeNew rcap- TextBuilder counters <$> newSTRef (TextArrays bytes bcap offsets val)---- | Append @len@ raw bytes at @off@ in @src@ as one field (one memcpy).-appendTextSlice :: TextBuilder s -> A.Array -> Int -> Int -> ST s ()-appendTextSlice b src off len = do- (n, pos, arrs) <- reserveText b len- A.copyI len (taBytes arrs) pos src off- finishTextAppend b arrs n (pos + len)-{-# INLINE appendTextSlice #-}---- | 'appendTextSlice' from foreign memory (e.g. an mmapped file buffer).-appendTextSliceFromPtr :: TextBuilder s -> Ptr Word8 -> Int -> ST s ()-appendTextSliceFromPtr b ptr len = do- (n, pos, arrs) <- reserveText b len- A.copyFromPointer (taBytes arrs) pos ptr len- finishTextAppend b arrs n (pos + len)-{-# INLINE appendTextSliceFromPtr #-}---- | Append an already-decoded 'Text' (its bytes are UTF-8 already).-appendText :: TextBuilder s -> T.Text -> ST s ()-appendText b (Text src off len) = appendTextSlice b src off len-{-# INLINE appendText #-}--finishTextAppend :: TextBuilder s -> TextArrays s -> Int -> Int -> ST s ()-finishTextAppend b arrs n endPos = do- anyNull <- VUM.unsafeRead (tbCounters b) cAnyNull- when (anyNull /= 0) $ VUM.unsafeWrite (taValid arrs) n 1- VUM.unsafeWrite (taOffsets arrs) (n + 1) endPos- VUM.unsafeWrite (tbCounters b) cRows (n + 1)- VUM.unsafeWrite (tbCounters b) cBytes endPos-{-# INLINE finishTextAppend #-}--reserveText :: TextBuilder s -> Int -> ST s (Int, Int, TextArrays s)-reserveText b extra = do- n <- VUM.unsafeRead (tbCounters b) cRows- pos <- VUM.unsafeRead (tbCounters b) cBytes- arrs <- readSTRef (tbArrays b)- arrs' <-- if n < VUM.length (taValid arrs) && pos + extra <= taByteCap arrs- then pure arrs- else growText b arrs (n + 1) (pos + extra)- pure (n, pos, arrs')-{-# INLINE reserveText #-}--growText :: TextBuilder s -> TextArrays s -> Int -> Int -> ST s (TextArrays s)-growText b (TextArrays bytes bcap offsets val) needRows needBytes = do- let rcap = VUM.length val- (offsets', val') <-- if needRows > rcap- then do- let rcap' = max (2 * rcap) needRows- o <- VUM.unsafeGrow offsets (rcap' - rcap)- v <- VUM.unsafeGrow val (rcap' - rcap)- pure (o, v)- else pure (offsets, val)- (bytes', bcap') <-- if needBytes > bcap- then do- let cap' = max (2 * bcap) needBytes- bs <- A.resizeM bytes cap'- pure (bs, cap')- else pure (bytes, bcap)- let arrs = TextArrays bytes' bcap' offsets' val'- writeSTRef (tbArrays b) arrs- pure arrs--{- | Freeze a 'TextBuilder' into a raw 'TextChunk' for byte-level merging-('mergeTextChunks'): no 'T.Text' values are created until chunks merge.--}-freezeTextChunk :: TextBuilder s -> ST s TextChunk-freezeTextChunk b = do- n <- VUM.unsafeRead (tbCounters b) cRows- anyNull <- VUM.unsafeRead (tbCounters b) cAnyNull- used <- VUM.unsafeRead (tbCounters b) cBytes- TextArrays bytes bcap offsets val <- readSTRef (tbArrays b)- when (used < bcap) (A.shrinkM bytes used)- arr <- A.unsafeFreeze bytes- offs <- VU.unsafeFreeze (VUM.slice 0 (n + 1) offsets)- bm <-- if anyNull /= 0- then Just <$> packValidity n val- else pure Nothing- pure (TextChunk arr used offs bm)--instance ColumnBuilder TextBuilder where- appendNull b = do- (n, pos, arrs) <- reserveText b 0- anyNull <- VUM.unsafeRead (tbCounters b) cAnyNull- when (anyNull == 0) $ do- VUM.set (VUM.slice 0 n (taValid arrs)) 1- VUM.unsafeWrite (tbCounters b) cAnyNull 1- VUM.unsafeWrite (taValid arrs) n 0- VUM.unsafeWrite (taOffsets arrs) (n + 1) pos- VUM.unsafeWrite (tbCounters b) cRows (n + 1)- VUM.unsafeWrite (tbCounters b) cBytes pos- {-# INLINE appendNull #-}-- builderLength b = VUM.unsafeRead (tbCounters b) cRows-- freezeBuilder b = do- chunk <- freezeTextChunk b- pure $! mergeTextChunks [chunk]
− src-internal/DataFrame/Internal/ColumnMerge.hs
@@ -1,185 +0,0 @@-{-# LANGUAGE BangPatterns #-}-{-# LANGUAGE FlexibleContexts #-}-{-# LANGUAGE GADTs #-}-{-# LANGUAGE ScopedTypeVariables #-}-{-# LANGUAGE TypeApplications #-}--{- | Concatenation of per-chunk 'Column's (e.g. from parallel CSV chunks). Text-columns merge at the byte level via 'TextChunk' \/ 'mergeTextChunks', so no-per-chunk 'Data.Text.Text' values are ever materialized.--}-module DataFrame.Internal.ColumnMerge (- TextChunk (..),- mergeColumns,- mergeTextChunks,- packedFromTextChunk,- packValidity,- spliceBitmaps,- tcRows,-) where--import qualified Data.Text.Array as A-import qualified Data.Vector as VB-import qualified Data.Vector.Unboxed as VU-import qualified Data.Vector.Unboxed.Mutable as VUM--import Control.Monad (foldM_, forM_, when)-import Control.Monad.ST (ST, runST)-import Data.Bits (shiftL, shiftR, (.&.), (.|.))-import Data.Maybe (fromMaybe, isNothing)-import Data.Type.Equality (testEquality, (:~:) (Refl))-import Data.Word (Word8)-import DataFrame.Internal.Column (- Bitmap,- Column (..),- Columnable,- allValidBitmap,- isMergedColumn,- isPackedText,- materializeMerged,- materializePacked,- )-import DataFrame.Internal.PackedText (mkPackedContiguous)-import Type.Reflection (typeRep)--{- | A frozen text-builder chunk: raw UTF-8 bytes plus row offsets (row @i@-spans bytes @[offsets!i, offsets!(i+1))@) and an optional validity bitmap.-'Data.Text.Text' values are only created when chunks merge into a 'Column'.--}-data TextChunk = TextChunk- { tcBytes :: !A.Array- , tcUsed :: !Int- , tcOffsets :: !(VU.Vector Int)- , tcBitmap :: !(Maybe Bitmap)- }--tcRows :: TextChunk -> Int-tcRows c = VU.length (tcOffsets c) - 1--{- | Freeze a builder chunk directly into a packed-text column: no-'Data.Text.Text' materialization, no UTF-8 validation pass (deferred to decode).-Not yet called by any reader.--}-packedFromTextChunk :: TextChunk -> Column-packedFromTextChunk (TextChunk arr _used offs bm) =- PackedText bm (mkPackedContiguous arr offs)--{- | Merge text chunks into one packed-text 'Column': one byte-array copy per-chunk, one offset rebase, then wrap the shared buffer + offsets as 'PackedText'-(no per-row header, decode deferred).--}-mergeTextChunks :: [TextChunk] -> Column-mergeTextChunks [] = error "DataFrame.Internal.ColumnMerge.mergeTextChunks: empty list"-mergeTextChunks [c] = packedFromTextChunk c-mergeTextChunks cs = runST $ do- let totalBytes = sum (map tcUsed cs)- totalRows = sum (map tcRows cs)- arr <- A.new (max 1 totalBytes)- offs <- VUM.unsafeNew (totalRows + 1)- VUM.unsafeWrite offs 0 0- let splice !byteBase !rowBase c = do- let n = tcRows c- co = tcOffsets c- A.copyI (tcUsed c) arr byteBase (tcBytes c) 0- forM_ [1 .. n] $ \i ->- VUM.unsafeWrite offs (rowBase + i) (byteBase + VU.unsafeIndex co i)- pure (byteBase + tcUsed c, rowBase + n)- foldM_ (\(b, r) c -> splice b r c) (0, 0) cs- farr <- A.unsafeFreeze arr- foffs <- VU.unsafeFreeze offs- let !bm = spliceBitmaps [(tcBitmap c, tcRows c) | c <- cs]- pure (PackedText bm (mkPackedContiguous farr foffs))--{- | Merge per-chunk columns into one column: one allocation + memcpy per-payload, with bitmaps spliced across non-byte-aligned chunk boundaries.-All chunks must have the same element type.--}-mergeColumns :: [Column] -> Column-mergeColumns [] = error "DataFrame.Internal.ColumnBuilder.mergeColumns: empty list"-mergeColumns [c] = c--- Normalize on the whole list, not the head: a packed or merged chunk in any--- position must demote every chunk to the common boxed form.-mergeColumns cols@(c0 : _)- | any isMergedColumn cols = mergeColumns (map materializeMerged cols)- | any isPackedText cols = mergeColumns (map materializePacked cols)-mergeColumns cols@(c0 : _) = case c0 of- PackedText _ _ -> mergeColumns (map materializePacked cols)- MergedColumn _ _ -> mergeColumns (map materializeMerged cols)- UnboxedColumn _ (_ :: VU.Vector a) ->- let parts = map (unboxedPart @a) cols- !merged = VU.concat (map snd parts)- !bm = spliceBitmaps [(mb, VU.length v) | (mb, v) <- parts]- in UnboxedColumn bm merged- BoxedColumn _ (_ :: VB.Vector a) ->- let parts = map (boxedPart @a) cols- !merged = VB.concat (map snd parts)- !bm = spliceBitmaps [(mb, VB.length v) | (mb, v) <- parts]- in BoxedColumn bm merged--unboxedPart ::- forall a. (Columnable a, VU.Unbox a) => Column -> (Maybe Bitmap, VU.Vector a)-unboxedPart (UnboxedColumn mb (v :: VU.Vector b)) =- case testEquality (typeRep @a) (typeRep @b) of- Just Refl -> (mb, v)- Nothing -> mergeMismatch-unboxedPart _ = mergeMismatch--boxedPart ::- forall a. (Columnable a) => Column -> (Maybe Bitmap, VB.Vector a)-boxedPart (BoxedColumn mb (v :: VB.Vector b)) =- case testEquality (typeRep @a) (typeRep @b) of- Just Refl -> (mb, v)- Nothing -> mergeMismatch-boxedPart _ = mergeMismatch--mergeMismatch :: a-mergeMismatch =- error "DataFrame.Internal.ColumnBuilder.mergeColumns: chunk column types differ"--{- | Splice chunk bitmaps end to end at the bit level. 'Nothing' if no chunk-carries a bitmap; chunks without one count as all-valid otherwise.--}-spliceBitmaps :: [(Maybe Bitmap, Int)] -> Maybe Bitmap-spliceBitmaps parts- | all (isNothing . fst) parts = Nothing- | otherwise = Just $ VU.create $ do- let total = sum (map snd parts)- outBytes = (total + 7) `shiftR` 3- mv <- VUM.replicate outBytes 0- let orInto i w =- when (i < outBytes && w /= 0) $ do- old <- VUM.unsafeRead mv i- VUM.unsafeWrite mv i (old .|. w)- splice !bitPos (mb, len) = do- let bm = fromMaybe (allValidBitmap len) mb- sh = bitPos .&. 7- byte0 = bitPos `shiftR` 3- lastIdx = ((len + 7) `shiftR` 3) - 1- tailBits = len .&. 7- lastMask =- if tailBits == 0 then 0xFF else (1 `shiftL` tailBits) - 1- forM_ [0 .. lastIdx] $ \k -> do- let raw = VU.unsafeIndex bm k- masked = if k == lastIdx then raw .&. lastMask else raw- w = fromIntegral masked :: Word- orInto (byte0 + k) (fromIntegral (w `shiftL` sh))- when (sh /= 0) $- orInto (byte0 + k + 1) (fromIntegral (w `shiftR` (8 - sh)))- pure (bitPos + len)- foldM_ splice 0 parts- pure mv---- | Pack a 0\/1 byte-per-row validity prefix into a bit-packed 'Bitmap'.-packValidity :: Int -> VUM.MVector s Word8 -> ST s Bitmap-packValidity n val = do- bytes <- VU.unsafeFreeze (VUM.slice 0 n val)- let assemble b =- let base = b `shiftL` 3- m = min 8 (n - base)- go !acc !k- | k >= m = acc- | VU.unsafeIndex bytes (base + k) /= 0 =- go (acc .|. (1 `shiftL` k)) (k + 1)- | otherwise = go acc (k + 1)- in go (0 :: Word8) 0- pure $! VU.generate ((n + 7) `shiftR` 3) assemble
+ src-internal/DataFrame/Internal/Control/Concurrent.hs view
@@ -0,0 +1,177 @@+{-# LANGUAGE BangPatterns #-}++-- | Shared concurrency primitives for the dataframe packages.+module DataFrame.Internal.Control.Concurrent (+ -- * Capabilities+ capabilities,+ capabilitiesIO,+ shouldParallelize,+ parThreshold,++ -- * Chunk planning (pure)+ splitChunkRange,+ chunksFor,+ boundsChunks,++ -- * Thread fan-out+ forkJoin,+ forkJoin_,++ -- * Chunked fan-out (per-chunk callbacks only)+ parallelChunks,+ parallelChunks_,+ parallelBounds_,++ -- * Work-stealing pools+ pooledIndices,+ pooledRun,+) where++import Control.Concurrent (forkFinally, getNumCapabilities)+import Control.Concurrent.MVar (newEmptyMVar, putMVar, takeMVar)+import Control.Exception (ErrorCall (..), SomeException, throwIO)+import Control.Monad (when)+import Data.IORef (atomicModifyIORef', newIORef)+import qualified Data.Vector as V+import qualified Data.Vector.Mutable as VM+import qualified Data.Vector.Unboxed as VU+import System.IO.Unsafe (unsafePerformIO)++capabilities :: Int+capabilities = unsafePerformIO getNumCapabilities+{-# NOINLINE capabilities #-}++capabilitiesIO :: IO Int+capabilitiesIO = getNumCapabilities+{-# INLINE capabilitiesIO #-}++shouldParallelize :: Int -> Int -> Bool+shouldParallelize threshold n = n >= threshold && capabilities > 1+{-# INLINE shouldParallelize #-}++{- | Row count below which a fan-out does not pay for itself, for the fixed-cost+per-row loops (grouping, the aggregation kernels). Pair it with+'shouldParallelize'. Kernels with a materially different per-row cost —+the join probe, the radix sort — set their own thresholds.+-}+parThreshold :: Int+parThreshold = 200000++splitChunkRange :: Int -> Int -> [(Int, Int)]+splitChunkRange k n+ | n <= 0 = []+ | otherwise =+ [ (lo, lo + len)+ | w <- [0 .. k' - 1]+ , let lo = w * q + min w r+ , let len = q + fromEnum (w < r)+ , len > 0+ ]+ where+ -- A non-positive width is a caller bug; clamp to one chunk rather than+ -- return [] and silently skip the rows.+ !k' = max 1 k+ (!q, !r) = n `quotRem` k'++chunksFor :: Int -> Int -> [(Int, Int)]+chunksFor !threshold !n+ | not (shouldParallelize threshold n) = [(0, n)]+ | otherwise = splitChunkRange capabilities n+{-# INLINE chunksFor #-}++{- | Adjacent pairs of a precomputed bounds vector of length @caps + 1@. Empty+ranges are NOT dropped: the bounds are the caller's and the slot count is+often meaningful. Endpoints are forced here, on the spawning thread, so a+worker never starts by evaluating an index thunk that retains @bs@.+-}+boundsChunks :: Int -> VU.Vector Int -> [(Int, Int)]+boundsChunks caps bs =+ [ (lo, hi)+ | w <- [0 .. caps - 1]+ , let !lo = VU.unsafeIndex bs w+ , let !hi = VU.unsafeIndex bs (w + 1)+ ]+{-# INLINE boundsChunks #-}++rethrow :: Either SomeException a -> IO a+rethrow = either throwIO pure+{-# INLINE rethrow #-}++forkJoin :: [IO a] -> IO [a]+forkJoin [] = pure []+forkJoin [act] = fmap (: []) act+forkJoin actions = do+ vars <- mapM spawn actions+ results <- mapM takeMVar vars+ mapM rethrow results+ where+ spawn act = do+ var <- newEmptyMVar+ _ <- forkFinally act (putMVar var)+ pure var+{-# INLINEABLE forkJoin #-}++forkJoin_ :: [IO ()] -> IO ()+forkJoin_ [] = pure ()+forkJoin_ [act] = act+forkJoin_ actions = do+ vars <- mapM spawn actions+ results <- mapM takeMVar vars+ mapM_ rethrow results+ where+ spawn act = do+ var <- newEmptyMVar+ _ <- forkFinally act (putMVar var)+ pure var+{-# INLINEABLE forkJoin_ #-}++parallelChunks :: Int -> Int -> (Int -> Int -> IO a) -> IO [a]+parallelChunks threshold n body =+ forkJoin [body lo hi | (!lo, !hi) <- chunksFor threshold n]+{-# NOINLINE parallelChunks #-} -- INLINE worsens performance here.++-- | 'parallelChunks' for chunk bodies run only for their effects.+parallelChunks_ :: Int -> Int -> (Int -> Int -> IO ()) -> IO ()+parallelChunks_ threshold n body =+ forkJoin_ [body lo hi | (!lo, !hi) <- chunksFor threshold n]+{-# INLINE parallelChunks_ #-}++parallelBounds_ :: Int -> VU.Vector Int -> (Int -> Int -> IO ()) -> IO ()+parallelBounds_ caps bs body =+ forkJoin_ [body lo hi | (!lo, !hi) <- boundsChunks caps bs]+{-# NOINLINE parallelBounds_ #-}++pooledIndices :: Int -> Int -> (Int -> IO ()) -> IO ()+pooledIndices width count body+ | count <= 0 = pure ()+ | width <= 1 = mapM_ body [0 .. count - 1]+ | otherwise = do+ next <- newIORef 0+ let worker = do+ i <- atomicModifyIORef' next (\j -> (j + 1, j))+ when (i < count) (body i >> worker)+ forkJoin_ (replicate (min width count) worker)+{-# NOINLINE pooledIndices #-}++pooledRun :: Int -> [IO a] -> IO [a]+pooledRun width actions+ | width >= n = forkJoin actions+ | otherwise = do+ next <- newIORef 0+ out <- VM.unsafeNew n+ acts <- VM.unsafeNew n+ sequence_ [VM.unsafeWrite acts i a | (i, a) <- zip [0 ..] actions]+ let worker = do+ i <- atomicModifyIORef' next (\j -> (j + 1, j))+ when (i < n) $ do+ act <- VM.unsafeRead acts i+ VM.unsafeWrite acts i consumed+ r <- act+ VM.write out i r+ worker+ forkJoin_ (replicate width worker)+ V.toList <$> V.freeze out+ where+ n = length actions+ consumed = throwIO (ErrorCall "pooledRun: slot already consumed")+{-# INLINEABLE pooledRun #-}
+ src-internal/DataFrame/Internal/Data/HashTable.hs view
@@ -0,0 +1,108 @@+{-# LANGUAGE BangPatterns #-}+{-# LANGUAGE RankNTypes #-}+{-# LANGUAGE ScopedTypeVariables #-}++{- | A flat, unboxed, open-addressing (linear-probe) hash table mapping a row's+key-hash to a dense group id, re-verifying the real key on every hash hit to+reject collisions. Runs in any 'PrimMonad' ('ST' for grouping, 'IO' per worker).+-}+module DataFrame.Internal.Data.HashTable (+ HashTable (..),+ newHashTable,+ htInsert,+ nextPow2Above,+) where++import Control.Monad.Primitive (PrimMonad, PrimState)+import Data.Bits (popCount, unsafeShiftR, (.&.))+import qualified Data.Vector.Unboxed.Mutable as VUM+import Data.Word (Word64)++{- | An open-addressing linear-probe table. @htMask@ is @capacity - 1@ (capacity+is a power of two) and maps a hash to its home slot.+-}+data HashTable s = HashTable+ { htHash :: !(VUM.MVector s Int)+ , htGroup :: !(VUM.MVector s Int)+ , htRep :: !(VUM.MVector s Int)+ , htMask :: !Int+ }++{- | Smallest power of two strictly greater than @n@, at least 2. Sizes the+table so the load factor stays below ~0.5 even when every row is a distinct+group.+-}+nextPow2Above :: Int -> Int+nextPow2Above n = go 2+ where+ go !p+ | p > n = p+ | otherwise = go (p * 2)+{-# INLINE nextPow2Above #-}++{- | Allocate an empty table able to hold up to @n@ distinct groups while+keeping the load factor under ~0.5 (capacity @= nextPow2Above (2*n)@). All+group slots start empty (@-1@).+-}+newHashTable :: (PrimMonad m) => Int -> m (HashTable (PrimState m))+newHashTable n = do+ let !cap = nextPow2Above (2 * max 1 n)+ h <- VUM.unsafeNew cap+ g <- VUM.replicate cap (-1)+ r <- VUM.unsafeNew cap+ pure (HashTable h g r (cap - 1))+{-# INLINE newHashTable #-}++{- | Look up @row@ (with precomputed @hash@) and return its dense group id: an+empty slot starts a new group via @nextGroup@, a stored-hash match is re-verified+with @eqRow@ before reuse. The 'Bool' is 'True' when a new group was created.+-}+htInsert ::+ (PrimMonad m) =>+ HashTable (PrimState m) ->+ -- | @eqRow a b@: do rows @a@ and @b@ have equal key columns?+ (Int -> Int -> Bool) ->+ -- | Next dense group id to assign if this row starts a new group.+ Int ->+ -- | Row index being inserted.+ Int ->+ -- | Precomputed hash of the row's key.+ Int ->+ m (Int, Bool)+htInsert ht eqRow nextGroup row hash = go (homeSlot mask hash)+ where+ !mask = htMask ht+ !hs = htHash ht+ !gs = htGroup ht+ !rs = htRep ht+ go !slot = do+ g <- VUM.unsafeRead gs slot+ if g < 0+ then do+ VUM.unsafeWrite hs slot hash+ VUM.unsafeWrite gs slot nextGroup+ VUM.unsafeWrite rs slot row+ pure (nextGroup, True)+ else do+ h <- VUM.unsafeRead hs slot+ if h == hash+ then do+ rep <- VUM.unsafeRead rs slot+ if eqRow rep row+ then pure (g, False)+ else go ((slot + 1) .&. mask)+ else go ((slot + 1) .&. mask)+{-# INLINE htInsert #-}++{- | Home slot of a hash: the top @log2 cap@ bits of a Fibonacci multiply. The+row hash's final FxHash step is a multiply, which leaves its LOW bits poorly+diffused — raw-text keys cluster into contiguous linear-probe pileups when+slotted by @hash .&. mask@, so the home slot must come from the top bits.+-}+homeSlot :: Int -> Int -> Int+homeSlot !mask !hash =+ fromIntegral+ ( (fromIntegral hash * (0x9E3779B97F4A7C15 :: Word64))+ `unsafeShiftR` (64 - popCount mask)+ )+{-# INLINE homeSlot #-}
+ src-internal/DataFrame/Internal/Data/PackedText.hs view
@@ -0,0 +1,361 @@+{-# LANGUAGE BangPatterns #-}++{- | Packed-text payload + byte-slice primitives. A 'PackedTextData' shares one+UTF-8 byte buffer across all rows of a string column, with @n+1@ row offsets, so+no per-row 'Data.Text.Text' header is materialized until decode is demanded.+Offsets and selection vectors are stored 'Int32' whenever their values fit+(Arrow-style), halving the per-row footprint of large string columns.+-}+module DataFrame.Internal.Data.PackedText (+ PackedTextData (..),+ PackedOffsets (..),+ PackedSel (..),+ offAt,+ offCount,+ selAt,+ selLength,+ mkPackedContiguous,+ mkPackedContiguous32,+ mkOffsets,+ mkSel,+ packedGather,+ packedTake,+ packedRowOffsets,+ packedLength,+ packedSlice,+ packedIndexText,+ sliceEqBytes,+ sliceCmpBytes,+) where++import qualified Data.Text as T+import qualified Data.Text.Array as A+import qualified Data.Vector.Unboxed as VU+import qualified Data.Vector.Unboxed.Mutable as VUM++import Data.Int (Int32)+import Data.Ord (comparing)+import Data.Text.Internal (Text (Text))+import DataFrame.Internal.Control.Concurrent (+ parThreshold,+ parallelChunks_,+ shouldParallelize,+ )+import DataFrame.Internal.Data.PackedText.Utf8 (+ isValidUtf8Slice,+ lenientDecodeSlice,+ )+import System.IO.Unsafe (unsafePerformIO)++{- | Row byte-offsets, physically 'Int32' when every value fits (total buffer+bytes < 2^31) and 'Int' otherwise. Values are non-negative byte positions.+-}+data PackedOffsets+ = Offs32 {-# UNPACK #-} !(VU.Vector Int32)+ | Offs64 {-# UNPACK #-} !(VU.Vector Int)++-- | Offset at index @i@, widened to 'Int'.+offAt :: PackedOffsets -> Int -> Int+offAt (Offs32 v) i = fromIntegral (VU.unsafeIndex v i)+offAt (Offs64 v) i = VU.unsafeIndex v i+{-# INLINE offAt #-}++-- | Number of offset entries (row count + 1).+offCount :: PackedOffsets -> Int+offCount (Offs32 v) = VU.length v+offCount (Offs64 v) = VU.length v+{-# INLINE offCount #-}++{- | A selection layer mapping logical rows to base rows; @-1@ marks an+invalid/null row. 'Int32' when the base row count fits.+-}+data PackedSel+ = Sel32 {-# UNPACK #-} !(VU.Vector Int32)+ | Sel64 {-# UNPACK #-} !(VU.Vector Int)++-- | Base row for logical row @i@ (may be @-1@).+selAt :: PackedSel -> Int -> Int+selAt (Sel32 v) i = fromIntegral (VU.unsafeIndex v i)+selAt (Sel64 v) i = VU.unsafeIndex v i+{-# INLINE selAt #-}++selLength :: PackedSel -> Int+selLength (Sel32 v) = VU.length v+selLength (Sel64 v) = VU.length v+{-# INLINE selLength #-}++{- | A shared UTF-8 byte buffer plus @n+1@ row offsets (base row @r@ spans bytes+@[offsets!r, offsets!(r+1))@); validity lives in the column's bitmap. @ptSel@ is+an optional selection layer letting a gather/join/sort result share the buffer.++@ptCanonicalSel@ marks a selection that is a canonical dictionary encoding:+equal byte slices always map to the same base row (codes). Set by dictionary+compaction; preserved by gather/take over an already-canonical selection (a+row keeps its code); 'False' for a gather over an unselected base, where two+logical rows can select different but equal-byted base rows. Grouping keys on+codes directly when it holds.+-}+data PackedTextData = PackedTextData+ { ptBytes :: {-# UNPACK #-} !A.Array+ , ptOffsets :: !PackedOffsets+ , ptSel :: !(Maybe PackedSel)+ , ptCanonicalSel :: !Bool+ }++int32Max :: Int+int32Max = fromIntegral (maxBound :: Int32)++-- | Narrow an 'Int' offset vector when the final offset (total bytes) fits.+mkOffsets :: VU.Vector Int -> PackedOffsets+mkOffsets offs+ | not (VU.null offs) && VU.last offs <= int32Max =+ Offs32 (VU.map fromIntegral offs)+ | otherwise = Offs64 offs+{-# INLINE mkOffsets #-}++{- | Narrow an 'Int' base-row vector (@-1@ sentinels allowed) when the base+row count fits in 'Int32'.+-}+mkSel :: Int -> VU.Vector Int -> PackedSel+mkSel base rows+ | base <= int32Max = Sel32 (VU.map fromIntegral rows)+ | otherwise = Sel64 rows+{-# INLINE mkSel #-}++-- | Build a contiguous packed payload (no selection): the freeze-path shape.+mkPackedContiguous :: A.Array -> VU.Vector Int -> PackedTextData+mkPackedContiguous arr offs = PackedTextData arr (mkOffsets offs) Nothing False+{-# INLINE mkPackedContiguous #-}++-- | 'mkPackedContiguous' from offsets already produced at 'Int32' width.+mkPackedContiguous32 :: A.Array -> VU.Vector Int32 -> PackedTextData+mkPackedContiguous32 arr offs = PackedTextData arr (Offs32 offs) Nothing False+{-# INLINE mkPackedContiguous32 #-}++{- | Reindex a packed payload by a selection vector, sharing the byte buffer;+logical row @i@ becomes base row @indices!i@. A negative or out-of-range index+decodes to the empty slice. Composes with an existing selection; canonicality+survives composition (a kept row keeps its code) but not a first selection+over the unselected base.+-}+packedGather :: VU.Vector Int -> PackedTextData -> PackedTextData+packedGather indices (PackedTextData arr offs msel canon) =+ let !base = offCount offs - 1+ canon' = case msel of+ Nothing -> False+ Just _ -> canon+ {- Base row for logical output row i composes any existing selection;+ only the SELECTOR is rebuilt (the byte buffer and offsets are shared+ untouched — a canonical dict column keeps its bytes). Each (source+ width x output width) shape runs a monomorphic closure-free kernel,+ generated directly at its final 'mkSel' width (one parallel pass).+ Same values and widths as the historical closure-driven build. -}+ sel' = case msel of+ Nothing+ | base <= int32Max -> Sel32 (gatherBaseTo32 indices base)+ | otherwise -> Sel64 (gatherBaseTo64 indices base)+ Just (Sel32 s)+ | base <= int32Max -> Sel32 (gatherSel32To32 indices s base)+ | otherwise -> Sel64 (gatherSel32To64 indices s base)+ Just (Sel64 s)+ | base <= int32Max -> Sel32 (gatherSel64To32 indices s base)+ | otherwise -> Sel64 (gatherSel64To64 indices s base)+ in PackedTextData arr offs (Just sel') canon'++-- Monomorphic 'packedGather' selector kernels. All reproduce exactly+-- @clamp r = if r >= 0 && r < base then r else -1@ over the composed pick.++gatherBaseTo32 :: VU.Vector Int -> Int -> VU.Vector Int32+gatherBaseTo32 indices !base =+ parGenSelInline (VU.length indices) $ \i ->+ let !r = VU.unsafeIndex indices i+ in if r >= 0 && r < base then fromIntegral r else -1+{-# NOINLINE gatherBaseTo32 #-}++gatherBaseTo64 :: VU.Vector Int -> Int -> VU.Vector Int+gatherBaseTo64 indices !base =+ parGenSelInline (VU.length indices) $ \i ->+ let !r = VU.unsafeIndex indices i+ in if r >= 0 && r < base then r else -1+{-# NOINLINE gatherBaseTo64 #-}++gatherSel32To32 :: VU.Vector Int -> VU.Vector Int32 -> Int -> VU.Vector Int32+gatherSel32To32 indices s !base =+ let !sn = VU.length s+ in parGenSelInline (VU.length indices) $ \i ->+ let !j = VU.unsafeIndex indices i+ in if j >= 0 && j < sn+ then+ let !r = fromIntegral (VU.unsafeIndex s j) :: Int+ in if r >= 0 && r < base then fromIntegral r else -1+ else -1+{-# NOINLINE gatherSel32To32 #-}++gatherSel32To64 :: VU.Vector Int -> VU.Vector Int32 -> Int -> VU.Vector Int+gatherSel32To64 indices s !base =+ let !sn = VU.length s+ in parGenSelInline (VU.length indices) $ \i ->+ let !j = VU.unsafeIndex indices i+ in if j >= 0 && j < sn+ then+ let !r = fromIntegral (VU.unsafeIndex s j) :: Int+ in if r >= 0 && r < base then r else -1+ else -1+{-# NOINLINE gatherSel32To64 #-}++gatherSel64To32 :: VU.Vector Int -> VU.Vector Int -> Int -> VU.Vector Int32+gatherSel64To32 indices s !base =+ let !sn = VU.length s+ in parGenSelInline (VU.length indices) $ \i ->+ let !j = VU.unsafeIndex indices i+ in if j >= 0 && j < sn+ then+ let !r = VU.unsafeIndex s j+ in if r >= 0 && r < base then fromIntegral r else -1+ else -1+{-# NOINLINE gatherSel64To32 #-}++gatherSel64To64 :: VU.Vector Int -> VU.Vector Int -> Int -> VU.Vector Int+gatherSel64To64 indices s !base =+ let !sn = VU.length s+ in parGenSelInline (VU.length indices) $ \i ->+ let !j = VU.unsafeIndex indices i+ in if j >= 0 && j < sn+ then+ let !r = VU.unsafeIndex s j+ in if r >= 0 && r < base then r else -1+ else -1+{-# NOINLINE gatherSel64To64 #-}++{- | Selector generate over 'parallelChunks_': one contiguous index chunk per+capability written into disjoint slices of one buffer — element @i@ depends+only on @f i@, so the result is bit-identical to 'VU.generate' at any @-N@.+Same policy as 'DataFrame.Internal.Column.Operations.parGenerateUnboxed', which+sits above this module.+-}++{- | 'parGenSel' with an INLINE body: each monomorphic NOINLINE kernel above+gets its own copy of the fill loop with the pick function inlined — no unknown+closure call (or boxed result allocation) per element. Bit-identical results;+callers must be NOINLINE so the 'unsafePerformIO' runs once per call.+-}+parGenSelInline :: (VU.Unbox c) => Int -> (Int -> c) -> VU.Vector c+parGenSelInline n f+ | not (shouldParallelize parThreshold n) = VU.generate n f+ | otherwise = unsafePerformIO $ do+ mv <- VUM.unsafeNew n+ parallelChunks_ parThreshold n $ \ !lo !hi ->+ let fill !i+ | i >= hi = pure ()+ | otherwise = VUM.unsafeWrite mv i (f i) >> fill (i + 1)+ in fill lo+ VU.unsafeFreeze mv+{-# INLINE parGenSelInline #-}++parGenSel :: (VU.Unbox c) => Int -> (Int -> c) -> VU.Vector c+{-# SPECIALIZE parGenSel :: Int -> (Int -> Int32) -> VU.Vector Int32 #-}+{-# SPECIALIZE parGenSel :: Int -> (Int -> Int) -> VU.Vector Int #-}+parGenSel n f+ | not (shouldParallelize parThreshold n) = VU.generate n f+ | otherwise = unsafePerformIO $ do+ mv <- VUM.unsafeNew n+ parallelChunks_ parThreshold n $ \ !lo !hi ->+ let fill !i+ | i >= hi = pure ()+ | otherwise = VUM.unsafeWrite mv i (f i) >> fill (i + 1)+ in fill lo+ VU.unsafeFreeze mv+{-# NOINLINE parGenSel #-}++{- | Take the first @k@ logical rows, sharing the byte buffer via a capped+selection layer. O(k), no byte copy or decode — cheap @take@/display on a+large packed column.+-}+packedTake :: Int -> PackedTextData -> PackedTextData+packedTake k (PackedTextData arr offs msel canon) =+ let !base = offCount offs - 1+ !k' = max 0 k+ (sel', canon') = case msel of+ Just (Sel32 s) -> (Sel32 (VU.take k' s), canon)+ Just (Sel64 s) -> (Sel64 (VU.take k' s), canon)+ Nothing -> (mkSel base (VU.enumFromN 0 (min k' base)), False)+ in PackedTextData arr offs (Just sel') canon'+{-# INLINE packedTake #-}++-- | Map a logical row index to its base row, honoring any selection layer.+baseRow :: PackedTextData -> Int -> Int+baseRow (PackedTextData _ _ Nothing _) i = i+baseRow (PackedTextData _ _ (Just sel) _) i = selAt sel i+{-# INLINE baseRow #-}++-- | Row count: @length sel@ when selected, else @length offsets - 1@.+packedLength :: PackedTextData -> Int+packedLength (PackedTextData _ offs Nothing _) = offCount offs - 1+packedLength (PackedTextData _ _ (Just sel) _) = selLength sel+{-# INLINE packedLength #-}++-- | Raw byte slice for logical row @i@: @(buffer, offset, length)@. The hot accessor.+packedSlice :: PackedTextData -> Int -> (A.Array, Int, Int)+packedSlice p@(PackedTextData arr offs _ _) i =+ let !r = baseRow p i+ in if r < 0+ then (arr, 0, 0)+ else+ let o = offAt offs r in (arr, o, offAt offs (r + 1) - o)+{-# INLINE packedSlice #-}++{- | The shared buffer + contiguous @n+1@ offsets when the payload is the+unselected base; a selected (gathered) payload returns 'Nothing' (its rows are+non-contiguous). Lets contiguous consumers skip the selection indirection.+-}+packedRowOffsets :: PackedTextData -> Maybe (A.Array, PackedOffsets)+packedRowOffsets (PackedTextData arr offs Nothing _) = Just (arr, offs)+packedRowOffsets _ = Nothing+{-# INLINE packedRowOffsets #-}++{- | On-demand single 'Data.Text.Text' for row @i@, using the same+validate-or-lenient decode as the freeze path so output is bit-identical.+-}+packedIndexText :: PackedTextData -> Int -> T.Text+packedIndexText p i =+ let (arr, o, l) = packedSlice p i+ in decodeField arr o l+{-# INLINE packedIndexText #-}++-- Decode one field exactly as the boxed freeze path does per row.+decodeField :: A.Array -> Int -> Int -> T.Text+decodeField arr o l+ | l == 0 = T.empty+ | isValidUtf8Slice arr o l = Text arr o l+ | otherwise = lenientDecodeSlice arr o l+{-# INLINE decodeField #-}++{- | Byte-wise equality of two slices. UTF-8 is injective on valid scalar+sequences and lenient decode is deterministic, so this agrees with+@Text@'s '==' on the decoded values.+-}+sliceEqBytes :: A.Array -> Int -> Int -> A.Array -> Int -> Int -> Bool+sliceEqBytes a ao al b bo bl+ | al /= bl = False+ | otherwise = go 0+ where+ go !k+ | k >= al = True+ | A.unsafeIndex a (ao + k) == A.unsafeIndex b (bo + k) = go (k + 1)+ | otherwise = False+{-# INLINE sliceEqBytes #-}++{- | Unsigned byte-lexicographic comparison (memcmp semantics). For+well-formed UTF-8 this matches 'Data.Text.compare' exactly, since UTF-8+byte order equals codepoint order for all valid scalars.+-}+sliceCmpBytes :: A.Array -> Int -> Int -> A.Array -> Int -> Int -> Ordering+sliceCmpBytes a ao al b bo bl = go 0+ where+ !m = min al bl+ go !k+ | k >= m = compare al bl+ | otherwise = case comparing id (A.unsafeIndex a (ao + k)) (A.unsafeIndex b (bo + k)) of+ EQ -> go (k + 1)+ r -> r+{-# INLINE sliceCmpBytes #-}
+ src-internal/DataFrame/Internal/Data/PackedText/Utf8.hs view
@@ -0,0 +1,95 @@+{-# LANGUAGE BangPatterns #-}++{- | UTF-8 validation and @decodeUtf8Lenient@-parity slice decoding used by+'DataFrame.Internal.Column.Builder' to turn shared byte buffers into 'Text'.+-}+module DataFrame.Internal.Data.PackedText.Utf8 (+ isValidUtf8Slice,+ isUtf8Boundary,+ lenientDecodeSlice,+ sliceTextVector,+) where++import qualified Data.Text as T+import qualified Data.Text.Array as A+import qualified Data.Vector as VB+import qualified Data.Vector.Mutable as VBM+import qualified Data.Vector.Unboxed as VU++import Data.Text.Internal (Text (..))+import Data.Text.Internal.Encoding.Utf8 (+ DecoderResult (..),+ utf8DecodeContinue,+ utf8DecodeStart,+ )+import Data.Text.Internal.Validate (isValidUtf8ByteArray)+import Data.Word (Word8)++-- | Whether @len@ bytes starting at @off@ are well-formed UTF-8.+isValidUtf8Slice :: A.Array -> Int -> Int -> Bool+isValidUtf8Slice = isValidUtf8ByteArray+{-# INLINE isValidUtf8Slice #-}++{- | Whether a byte may start a code point (i.e. is not a continuation+byte). Field slices of a valid buffer are themselves valid iff every+field starts on a boundary.+-}+isUtf8Boundary :: Word8 -> Bool+isUtf8Boundary w = w < 0x80 || w >= 0xC0+{-# INLINE isUtf8Boundary #-}++{- | Decode a byte slice exactly like @decodeUtf8Lenient@: greedy decode at+each position; any byte that cannot begin a complete, valid sequence within+the slice becomes one U+FFFD and decoding resumes at the next byte.+-}+lenientDecodeSlice :: A.Array -> Int -> Int -> T.Text+lenientDecodeSlice arr off len = T.pack (go off)+ where+ !end = off + len+ go !i+ | i >= end = []+ | otherwise = case tryDecode i of+ Just (c, i') -> c : go i'+ Nothing -> '\xFFFD' : go (i + 1)+ tryDecode !i = loop (utf8DecodeStart (A.unsafeIndex arr i)) (i + 1)+ where+ loop (Accept c) !j = Just (c, j)+ loop Reject _ = Nothing+ loop (Incomplete st cp) !j+ | j >= end = Nothing+ | otherwise = loop (utf8DecodeContinue (A.unsafeIndex arr j) st cp) (j + 1)++{- | Slice forced 'Text' values off a shared array; row @i@ spans bytes+@[offs!i, offs!(i+1))@. Fast path validates the whole span once when every field+starts on a code-point boundary; else per-field validation with lenient decode.+-}+sliceTextVector :: A.Array -> VU.Vector Int -> VB.Vector T.Text+sliceTextVector arr offs = VB.create $ do+ mv <- VBM.unsafeNew n+ let fill dec = go 0+ where+ go !i+ | i >= n = pure ()+ | otherwise = do+ let o = VU.unsafeIndex offs i+ !t = dec o (VU.unsafeIndex offs (i + 1) - o)+ VBM.unsafeWrite mv i t+ go (i + 1)+ if fast then fill mkSlice else fill decodeField+ pure mv+ where+ n = VU.length offs - 1+ base = VU.unsafeIndex offs 0+ used = VU.unsafeIndex offs n+ boundariesOk !i+ | i >= n = True+ | otherwise =+ let o = VU.unsafeIndex offs i+ in (o >= used || isUtf8Boundary (A.unsafeIndex arr o))+ && boundariesOk (i + 1)+ fast = isValidUtf8Slice arr base (used - base) && boundariesOk 0+ mkSlice o l = if l == 0 then T.empty else Text arr o l+ decodeField o l+ | l == 0 = T.empty+ | isValidUtf8Slice arr o l = Text arr o l+ | otherwise = lenientDecodeSlice arr o l
src-internal/DataFrame/Internal/DataFrame.hs view
@@ -8,7 +8,34 @@ {-# LANGUAGE ScopedTypeVariables #-} {-# LANGUAGE TypeApplications #-} -module DataFrame.Internal.DataFrame where+module DataFrame.Internal.DataFrame (+ DataFrame (..),+ forceDataFrame,+ GroupedDataFrame (.., Grouped),+ TruncateConfig (..),+ defaultTruncateConfig,+ ellipsisText,+ toMarkdown,+ toMarkdown',+ asText,+ asTextWith,+ pickColumns,+ insertAt,+ truncateCell,+ empty,+ columnNames,+ insertColumn,+ fromNamedColumns,+ getColumn,+ unsafeGetColumn,+ null,+ toCsv,+ toCsv',+ toSeparated,+ getRowAsText,+ showElement,+ stripJust,+) where import qualified Data.Map as M import qualified Data.Text as T@@ -24,13 +51,30 @@ type (:~:) (Refl), type (:~~:) (HRefl), )-import DataFrame.Display.Terminal.PrettyPrint-import DataFrame.Errors-import DataFrame.Internal.Column+import DataFrame.Display.Terminal.PrettyPrint (+ RenderFormat (..),+ showTable,+ )+import DataFrame.Errors (+ DataFrameException (ColumnsNotFoundException),+ )+import DataFrame.Internal.Column (+ Column (..),+ columnLength,+ columnToTextVec,+ columnTypeString,+ expandColumn,+ forceColumn,+ materializeMerged,+ parBackpermute2Int,+ sliceColumn,+ takeColumn,+ )+import DataFrame.Internal.Column.Bitmap (bitmapTestBit)+import DataFrame.Internal.Data.PackedText (packedIndexText) import DataFrame.Internal.Expression-import DataFrame.Internal.PackedText (packedIndexText)-import Text.Printf-import Type.Reflection (Typeable, eqTypeRep, typeRep, pattern App)+import Text.Printf (printf)+import Type.Reflection (eqTypeRep, typeRep, pattern App) import Prelude hiding (null) data DataFrame = DataFrame@@ -53,18 +97,64 @@ {- | A record that contains information about how and what rows are grouped in the dataframe. This can only be used with `aggregate`.++Laziness contract: every field except 'valueIndices' and 'groupRepRows' is+computed eagerly by the grouping paths. 'valueIndices' may be a lazy thunk (the+low-cardinality direct grouping defers the O(n) stable placement pass until a+consumer — grouped median/top-k gathers, set ops, the interpreter's+group-slicing — actually demands the permutation); forcing it always yields the+unique stable counting-sort permutation of 'rowToGroup', so WHAT it evaluates to+is independent of when it is forced. 'groupRepRows' is the per-group+representative row (the first original row of each group, in canonical group+order); aggregation uses it to materialize the key columns without demanding+'valueIndices'. It may also be a thunk; its value always equals+@VU.map (valueIndices !) (VU.init offsets)@. -}-data GroupedDataFrame = Grouped+data GroupedDataFrame = GroupedInternal { fullDataframe :: DataFrame , groupedColumns :: [T.Text] , valueIndices :: VU.Vector Int+ -- ^ Rows sorted by group id (stable); possibly an unevaluated thunk. , offsets :: VU.Vector Int , rowToGroup :: VU.Vector Int {- ^ rowToGroup[i] = group index for row i. Length n (one per row). Built once in 'groupBy'; reused by every aggregation. -}+ , groupRepRows :: VU.Vector Int+ -- ^ First original row of each group, length nGroups. See laziness note. } +{- | The historical five-field view of 'GroupedDataFrame'. Matching ignores+'groupRepRows'; building derives it lazily from @valueIndices@/@offsets@ (the+thunk only forces them if something actually reads the representative rows).+The direct grouping paths construct 'GroupedInternal' directly instead so the+representative rows never demand the placement pass.+-}+pattern Grouped ::+ DataFrame ->+ [T.Text] ->+ VU.Vector Int ->+ VU.Vector Int ->+ VU.Vector Int ->+ GroupedDataFrame+pattern Grouped df cols vis offs rtg <- GroupedInternal df cols vis offs rtg _+ where+ Grouped df cols vis offs rtg =+ GroupedInternal+ df+ cols+ vis+ offs+ rtg+ {- Parallel gather (still a deferred thunk; forcing it forces+ vis/offs as before). Indices are grouping-produced and+ in-bounds by construction: offs has nGroups+1 entries and+ offs!g < length vis for every non-empty group. Values are+ identical to the historical @VU.map (vis !) (VU.init offs)@. -}+ (parBackpermute2Int vis offs)++{-# COMPLETE Grouped #-}+ instance Show GroupedDataFrame where show (Grouped df cols _indices _os _rtg) = printf@@ -150,8 +240,8 @@ (takeColumn rowCap) ((V.!?) (columns d) ((M.!) (columnIndices d) name)) survivingCols = map lookupCol visibleHeaders- survivingTypes = map (maybe "" getType) survivingCols- survivingData = map get survivingCols+ survivingTypes = map (maybe "" (T.pack . columnTypeString)) survivingCols+ survivingData = map (maybe V.empty columnToTextVec) survivingCols clipCell = case mTrunc of Just cfg | maxCellWidth cfg > 0 -> truncateCell (maxCellWidth cfg)@@ -165,42 +255,6 @@ , insertAt i ellipsisText survivingTypes , insertAt i ellipsisCol survivingData )-- getType :: Column -> T.Text- showMaybeType :: forall a. (Typeable a) => String- showMaybeType =- let s = show (typeRep @a)- in "Maybe " <> if ' ' `elem` s then "(" <> s <> ")" else s- getType (BoxedColumn Nothing (_ :: V.Vector a)) = T.pack $ show (typeRep @a)- getType (BoxedColumn (Just _) (_ :: V.Vector a)) = T.pack $ showMaybeType @a- getType (UnboxedColumn Nothing (_ :: VU.Vector a)) = T.pack $ show (typeRep @a)- getType (UnboxedColumn (Just _) (_ :: VU.Vector a)) = T.pack $ showMaybeType @a- getType (PackedText Nothing _) = T.pack $ show (typeRep @T.Text)- getType (PackedText (Just _) _) = T.pack $ showMaybeType @T.Text- getType c@(MergedColumn _ _) = getType (mergedHead c)-- get :: Maybe Column -> V.Vector T.Text- get (Just c@(MergedColumn _ _)) = get (Just (materializeMerged c))- get (Just (BoxedColumn (Just bm) (column :: V.Vector a))) =- V.generate (V.length column) $ \i ->- if bitmapTestBit bm i- then T.pack (show (Just (V.unsafeIndex column i)))- else "Nothing"- get (Just (BoxedColumn Nothing (column :: V.Vector a))) =- case testEquality (typeRep @a) (typeRep @T.Text) of- Just Refl -> column- Nothing -> case testEquality (typeRep @a) (typeRep @String) of- Just Refl -> V.map T.pack column- Nothing -> V.map (T.pack . show) column- get (Just (UnboxedColumn (Just bm) column)) =- V.generate (VU.length column) $ \i ->- if bitmapTestBit bm i- then T.pack (show (Just (VU.unsafeIndex column i)))- else "Nothing"- get (Just (UnboxedColumn Nothing column)) =- V.generate (VU.length column) (T.pack . show . VU.unsafeIndex column)- get (Just c@(PackedText _ _)) = get (Just (materializePacked c))- get Nothing = V.empty in showTable fmt (map clipCell finalHeaders)
− src-internal/DataFrame/Internal/DictEncode.hs
@@ -1,171 +0,0 @@-{-# LANGUAGE BangPatterns #-}-{-# LANGUAGE ExplicitNamespaces #-}-{-# LANGUAGE GADTs #-}-{-# LANGUAGE ScopedTypeVariables #-}-{-# LANGUAGE TypeApplications #-}--{- | Dictionary-encode a text (or factor) group key to dense @Int@ codes: each row-gets a first-appearance code @0..card-1@ (NULL reserved) plus the cardinality. A-tested building block; profiled slower than the hash group-by, so unused for now.--}-module DataFrame.Internal.DictEncode (- dictEncodeColumn,- dictEncodeColumnUpTo,- dictCompactColumn,- dictMaxCardinality,-) where--import Control.Monad (when)-import Control.Monad.ST (runST)-import qualified Data.Text as T-import qualified Data.Text.Array as A-import Data.Type.Equality (TestEquality (..), type (:~:) (Refl))-import qualified Data.Vector as V-import qualified Data.Vector.Unboxed as VU-import qualified Data.Vector.Unboxed.Mutable as VUM-import Type.Reflection (typeRep)--import DataFrame.Internal.Column (Bitmap, Column (..), bitmapTestBit)-import DataFrame.Internal.Hash (fnvOffset, mixBytes, mixText, nullSalt)-import DataFrame.Internal.HashTable (htInsert, newHashTable)-import DataFrame.Internal.PackedText (- PackedTextData (..),- mkOffsets,- mkSel,- packedLength,- packedSlice,- sliceEqBytes,- )--{- | Largest distinct-value count we will dictionary-encode. Above this the codes-no longer index a reasonable direct accumulator and the encode pass is pure-overhead, so the caller keeps the plain hash group-by.--}-dictMaxCardinality :: Int-dictMaxCardinality = 1048576--{- | Dictionary-encode a text-like column to dense first-appearance @Int@ codes,-returning @Just (codes, cardinality)@ (a NULL row gets its own reserved code).-'Nothing' for non-text columns or cardinality above 'dictMaxCardinality'.--}-dictEncodeColumn :: Column -> Maybe (VU.Vector Int, Int)-dictEncodeColumn = dictEncodeColumnUpTo dictMaxCardinality--{- | Dictionary-encode like 'dictEncodeColumn' but bail to 'Nothing' as soon as-the distinct count would exceed @maxCard@, letting a low-cardinality probe avoid-a full high-cardinality pass.--}-dictEncodeColumnUpTo :: Int -> Column -> Maybe (VU.Vector Int, Int)-dictEncodeColumnUpTo maxCard (PackedText bm p) = encodePacked maxCard bm p-dictEncodeColumnUpTo maxCard (BoxedColumn bm (v :: V.Vector a)) =- case testEquality (typeRep @a) (typeRep @T.Text) of- Just Refl -> encodeBoxedText maxCard bm v- Nothing -> Nothing-dictEncodeColumnUpTo _ _ = Nothing--{- | Encode a packed-text column: hash each row's raw UTF-8 bytes (the grouping-'mixBytes'), re-verify byte equality on collisions, assign dense codes in-first-appearance order. A null row hashes 'nullSalt'.--}-encodePacked ::- Int -> Maybe Bitmap -> PackedTextData -> Maybe (VU.Vector Int, Int)-encodePacked maxCard bm p =- let !n = packedLength p- valid i = case bm of- Just b -> bitmapTestBit b i- Nothing -> True- hashAt i =- if valid i- then let (arr, o, l) = packedSlice p i in mixBytes fnvOffset arr o l- else nullSalt- eqAt a b =- case (valid a, valid b) of- (True, True) ->- let (arrA, oA, lA) = packedSlice p a- (arrB, oB, lB) = packedSlice p b- in sliceEqBytes arrA oA lA arrB oB lB- (False, False) -> True- _ -> False- in buildCodes maxCard n hashAt eqAt--{- | Encode a boxed 'Data.Text.Text' column, mirroring 'encodePacked' but over-boxed values (used when a user-built Text column is grouped).--}-encodeBoxedText ::- Int -> Maybe Bitmap -> V.Vector T.Text -> Maybe (VU.Vector Int, Int)-encodeBoxedText maxCard bm v =- let !n = V.length v- valid i = case bm of- Just b -> bitmapTestBit b i- Nothing -> True- hashAt i =- if valid i then mixText fnvOffset (V.unsafeIndex v i) else nullSalt- eqAt a b =- case (valid a, valid b) of- (True, True) -> V.unsafeIndex v a == V.unsafeIndex v b- (False, False) -> True- _ -> False- in buildCodes maxCard n hashAt eqAt--{- | The shared code-assignment loop: bucket every row through an open-addressing-table on its precomputed hash, re-verify with @eqAt@ on a hit, assign dense-first-appearance codes. Bails to 'Nothing' once the distinct count exceeds @maxCard@.--}-buildCodes ::- Int -> Int -> (Int -> Int) -> (Int -> Int -> Bool) -> Maybe (VU.Vector Int, Int)-buildCodes maxCard n hashAt eqAt- | n == 0 = Just (VU.empty, 0)- | otherwise = runST $ do- ht <- newHashTable (min n (maxCard + 1))- codes <- VUM.new n- let go !i !next- | i >= n = pure (Just next)- | next > maxCard = pure Nothing- | otherwise = do- let !h = hashAt i- (code, isNew) <- htInsert ht eqAt next i h- VUM.unsafeWrite codes i code- go (i + 1) (if isNew then next + 1 else next)- mres <- go 0 0- case mres of- Nothing -> pure Nothing- Just card -> do- frozen <- VU.unsafeFreeze codes- pure (Just (frozen, card))--dictCompactColumn :: Column -> Column-dictCompactColumn col@(PackedText bm p) =- case encodePacked dictMaxCardinality bm p of- Just (codes, card)- | 2 * card <= packedLength p ->- PackedText bm (dictPacked p codes card)- _ -> col-dictCompactColumn col = col--dictPacked :: PackedTextData -> VU.Vector Int -> Int -> PackedTextData-dictPacked p codes card = runST $ do- let n = VU.length codes- reps <- VUM.replicate card (-1)- let findReps !i !remaining- | remaining <= 0 || i >= n = pure ()- | otherwise = do- let c = VU.unsafeIndex codes i- cur <- VUM.unsafeRead reps c- if cur < 0- then VUM.unsafeWrite reps c i >> findReps (i + 1) (remaining - 1)- else findReps (i + 1) remaining- findReps 0 card- repsV <- VU.unsafeFreeze reps- let lens = VU.map (\r -> let (_, _, l) = packedSlice p r in l) repsV- offs = VU.scanl' (+) 0 lens- total = VU.last offs- marr <- A.new (max 1 total)- let copyRep !c =- when (c < card) $ do- let r = VU.unsafeIndex repsV c- (arr, o, l) = packedSlice p r- A.copyI l marr (VU.unsafeIndex offs c) arr o- copyRep (c + 1)- copyRep 0- arr <- A.unsafeFreeze marr- pure (PackedTextData arr (mkOffsets offs) (Just (mkSel card codes)) True)
+ src-internal/DataFrame/Internal/Display/Pretty.hs view
@@ -0,0 +1,129 @@+{- | A minimal Wadler/Leijen-style document combinator and width-aware renderer.+A 'Doc' describes a layout abstractly; 'render' chooses where soft breaks become+newlines to fit a target width. 'Group' lays a region flat when it fits.+-}+module DataFrame.Internal.Display.Pretty (+ Doc,+ text,+ line,+ hardline,+ nest,+ group,+ (<+>),+ hcat,+ punctuate,+ parens,+ parensWhenBroken,+ defaultWidth,+ render,+) where++data Doc+ = Empty+ | Text String+ | Line+ | Cat Doc Doc+ | Nest Int Doc+ | Group Doc+ | Hard+ | Alt Doc Doc++instance Semigroup Doc where+ (<>) = Cat++instance Monoid Doc where+ mempty = Empty++-- | A literal chunk of text. Must not contain newlines (use 'line'/'hardline').+text :: String -> Doc+text = Text++{- | A soft break: a single space when its enclosing 'group' fits the width,+otherwise a newline + current indentation.+-}+line :: Doc+line = Line++-- | A hard break that never flattens; any enclosing 'group' is forced to break.+hardline :: Doc+hardline = Hard++-- | Add @k@ spaces to the indentation applied at line breaks inside @d@.+nest :: Int -> Doc -> Doc+nest = Nest++-- | Lay the document out flat if it fits the remaining width, broken otherwise.+group :: Doc -> Doc+group = Group++-- | Concatenate two documents separated by a single space.+(<+>) :: Doc -> Doc -> Doc+x <+> y = x <> Text " " <> y++infixr 6 <+>++hcat :: [Doc] -> Doc+hcat = mconcat++-- | Append @sep@ after every element but the last.+punctuate :: Doc -> [Doc] -> [Doc]+punctuate _ [] = []+punctuate _ [d] = [d]+punctuate sep (d : ds) = (d <> sep) : punctuate sep ds++parens :: Doc -> Doc+parens d = Text "(" <> d <> Text ")"++{- | Render @d@ bare when it fits flat on the current line, wrapped in parens when+it must break across lines. Keeps operator grouping unambiguous once a+sub-expression wraps, without parenthesis noise on one-line expressions.+-}+parensWhenBroken :: Doc -> Doc+parensWhenBroken d = Group (Alt d (parens d))++defaultWidth :: Int+defaultWidth = 80++data Mode = Flat | Break++-- | Render a document, breaking soft lines so output fits @width@ columns.+render :: Int -> Doc -> String+render width doc = layout 0 [(0, Break, doc)]+ where+ layout :: Int -> [(Int, Mode, Doc)] -> String+ layout _ [] = ""+ layout col ((i, m, d) : rest) = case d of+ Empty -> layout col rest+ Text s -> s ++ layout (col + length s) rest+ Cat x y -> layout col ((i, m, x) : (i, m, y) : rest)+ Nest j x -> layout col ((i + j, m, x) : rest)+ Line -> case m of+ Flat -> ' ' : layout (col + 1) rest+ Break -> '\n' : replicate i ' ' ++ layout i rest+ Hard -> '\n' : replicate i ' ' ++ layout i rest+ Group x ->+ if fits (width - col) ((i, Flat, x) : rest)+ then layout col ((i, Flat, x) : rest)+ else layout col ((i, Break, x) : rest)+ Alt flat broken -> case m of+ Flat -> layout col ((i, Flat, flat) : rest)+ Break -> layout col ((i, Break, broken) : rest)++ fits :: Int -> [(Int, Mode, Doc)] -> Bool+ fits w _ | w < 0 = False+ fits _ [] = True+ fits w ((i, m, d) : rest) = case d of+ Empty -> fits w rest+ Text s -> fits (w - length s) rest+ Cat x y -> fits w ((i, m, x) : (i, m, y) : rest)+ Nest j x -> fits w ((i + j, m, x) : rest)+ Line -> case m of+ Flat -> fits (w - 1) rest+ Break -> True+ Hard -> case m of+ Flat -> False+ Break -> True+ Group x -> fits w ((i, Flat, x) : rest)+ Alt flat broken -> case m of+ Flat -> fits w ((i, Flat, flat) : rest)+ Break -> fits w ((i, Break, broken) : rest)
src-internal/DataFrame/Internal/Expression.hs view
@@ -22,7 +22,7 @@ import Data.Type.Equality (TestEquality (testEquality), type (:~:) (Refl)) import qualified Data.Vector.Generic as VG import DataFrame.Internal.Column-import qualified DataFrame.Internal.Pretty as P+import qualified DataFrame.Internal.Display.Pretty as P import Type.Reflection (Typeable, typeOf, typeRep) {- | Operators are an open typeclass: built-ins get their own 'Typeable' type so the
+ src-internal/DataFrame/Internal/Expression/Operators.hs view
@@ -0,0 +1,425 @@+{-# LANGUAGE FlexibleContexts #-}+{-# LANGUAGE GADTs #-}+{-# LANGUAGE OverloadedStrings #-}+{-# LANGUAGE TypeFamilies #-}+{-# LANGUAGE TypeOperators #-}++module DataFrame.Internal.Expression.Operators where++import Data.Function ((&))+import qualified Data.Text as T+import DataFrame.Internal.Column (Columnable)+import DataFrame.Internal.Column.Types (Promote, PromoteDiv)+import DataFrame.Internal.Expression (+ BinUDF (MkBinaryOp),+ BinaryOp (+ binaryCommutative,+ binaryFn,+ binaryName,+ binaryPrecedence,+ binarySymbol+ ),+ Expr (Binary, Col, If, Lit, Unary),+ NamedExpr,+ UExpr (UExpr),+ UnUDF (MkUnaryOp),+ )+import DataFrame.Internal.Expression.Operators.Nullable (+ BaseType,+ DivWidenOp,+ NullCmpResult,+ NullLift2Op (applyNull2),+ NullableCmpOp (nullCmpOp),+ NumericWidenOp,+ WidenResult,+ WidenResultDiv,+ divArithOp,+ widenArithOp,+ widenCmpOp,+ )++infixr 8 .^^, .^^., .^, .^.+infixl 7 .*, ./, .*., ./.+infixl 6 .+, .-, .+., .-.+infix 4 .==, .==., .<, .<., .<=, .<=., .>=, .>=., .>, .>., ./=, ./=.+infixr 3 .&&, .&&.+infixr 2 .||, .||.+infixr 0 .=++(|>) :: a -> (a -> b) -> b+(|>) = (&)++as :: (Columnable a) => Expr a -> T.Text -> NamedExpr+as expr colName = (colName, UExpr expr)++name :: (Show a) => Expr a -> T.Text+name (Col n) = n+name other =+ error $+ "You must call `name` on a column reference. Not the expression: " ++ show other++col :: (Columnable a) => T.Text -> Expr a+col = Col++ifThenElse :: (Columnable a) => Expr Bool -> Expr a -> Expr a -> Expr a+ifThenElse = If++lit :: (Columnable a) => a -> Expr a+lit = Lit++(.=) :: (Columnable a) => T.Text -> Expr a -> NamedExpr+(.=) = flip as++liftDecorated ::+ (Columnable a, Columnable b) =>+ (a -> b) -> T.Text -> Maybe T.Text -> Expr a -> Expr b+liftDecorated f opName rep = Unary (MkUnaryOp f opName rep)++lift2Decorated ::+ (Columnable c, Columnable b, Columnable a) =>+ (c -> b -> a) ->+ T.Text ->+ Maybe T.Text ->+ Bool ->+ Int ->+ Expr c ->+ Expr b ->+ Expr a+lift2Decorated f opName rep comm prec =+ Binary (MkBinaryOp f opName rep comm prec)++data NullEq a b c where+ NullEq ::+ ( NumericWidenOp (BaseType a) (BaseType b)+ , NullLift2Op a b Bool (NullCmpResult a b)+ , Eq (Promote (BaseType a) (BaseType b))+ ) =>+ NullEq a b (NullCmpResult a b)++data NullNeq a b c where+ NullNeq ::+ ( NumericWidenOp (BaseType a) (BaseType b)+ , NullLift2Op a b Bool (NullCmpResult a b)+ , Eq (Promote (BaseType a) (BaseType b))+ ) =>+ NullNeq a b (NullCmpResult a b)++data NullLt a b c where+ NullLt ::+ ( NumericWidenOp (BaseType a) (BaseType b)+ , NullLift2Op a b Bool (NullCmpResult a b)+ , Ord (Promote (BaseType a) (BaseType b))+ ) =>+ NullLt a b (NullCmpResult a b)++data NullGt a b c where+ NullGt ::+ ( NumericWidenOp (BaseType a) (BaseType b)+ , NullLift2Op a b Bool (NullCmpResult a b)+ , Ord (Promote (BaseType a) (BaseType b))+ ) =>+ NullGt a b (NullCmpResult a b)++data NullLeq a b c where+ NullLeq ::+ ( NumericWidenOp (BaseType a) (BaseType b)+ , NullLift2Op a b Bool (NullCmpResult a b)+ , Ord (Promote (BaseType a) (BaseType b))+ ) =>+ NullLeq a b (NullCmpResult a b)++data NullGeq a b c where+ NullGeq ::+ ( NumericWidenOp (BaseType a) (BaseType b)+ , NullLift2Op a b Bool (NullCmpResult a b)+ , Ord (Promote (BaseType a) (BaseType b))+ ) =>+ NullGeq a b (NullCmpResult a b)++data NullAnd a b c where+ NullAnd ::+ (NullableCmpOp a b (NullCmpResult a b), BaseType a ~ Bool) =>+ NullAnd a b (NullCmpResult a b)++data NullOr a b c where+ NullOr ::+ (NullableCmpOp a b (NullCmpResult a b), BaseType a ~ Bool) =>+ NullOr a b (NullCmpResult a b)++instance BinaryOp NullEq where+ binaryFn NullEq = applyNull2 (widenCmpOp (==))+ binaryName NullEq = "eq"+ binarySymbol NullEq = Just ".=="+ binaryCommutative NullEq = True+ binaryPrecedence NullEq = 4+instance BinaryOp NullNeq where+ binaryFn NullNeq = applyNull2 (widenCmpOp (/=))+ binaryName NullNeq = "neq"+ binarySymbol NullNeq = Just "./="+ binaryCommutative NullNeq = True+ binaryPrecedence NullNeq = 4+instance BinaryOp NullLt where+ binaryFn NullLt = applyNull2 (widenCmpOp (<))+ binaryName NullLt = "lt"+ binarySymbol NullLt = Just ".<"+ binaryPrecedence NullLt = 4+instance BinaryOp NullGt where+ binaryFn NullGt = applyNull2 (widenCmpOp (>))+ binaryName NullGt = "gt"+ binarySymbol NullGt = Just ".>"+ binaryPrecedence NullGt = 4+instance BinaryOp NullLeq where+ binaryFn NullLeq = applyNull2 (widenCmpOp (<=))+ binaryName NullLeq = "leq"+ binarySymbol NullLeq = Just ".<="+ binaryPrecedence NullLeq = 4+instance BinaryOp NullGeq where+ binaryFn NullGeq = applyNull2 (widenCmpOp (>=))+ binaryName NullGeq = "geq"+ binarySymbol NullGeq = Just ".>="+ binaryPrecedence NullGeq = 4+instance BinaryOp NullAnd where+ binaryFn NullAnd = nullCmpOp (&&)+ binaryName NullAnd = "nulland"+ binarySymbol NullAnd = Just ".&&"+ binaryCommutative NullAnd = True+ binaryPrecedence NullAnd = 3+instance BinaryOp NullOr where+ binaryFn NullOr = nullCmpOp (||)+ binaryName NullOr = "nullor"+ binarySymbol NullOr = Just ".||"+ binaryCommutative NullOr = True+ binaryPrecedence NullOr = 2++(.==.) ::+ (Columnable a, Eq a) =>+ Expr a ->+ Expr a ->+ Expr Bool+(.==.) = lift2Decorated (==) "eq" (Just ".==.") True 4++(./=.) ::+ (Columnable a, Eq a) =>+ Expr a ->+ Expr a ->+ Expr Bool+(./=.) = lift2Decorated (/=) "neq" (Just "./=.") True 4++(.<.) ::+ (Columnable a, Ord a) =>+ Expr a ->+ Expr a ->+ Expr Bool+(.<.) = lift2Decorated (<) "lt" (Just ".<.") False 4++(.>.) ::+ (Columnable a, Ord a) =>+ Expr a ->+ Expr a ->+ Expr Bool+(.>.) = lift2Decorated (>) "gt" (Just ".>.") False 4++(.<=.) ::+ (Columnable a, Ord a) =>+ Expr a ->+ Expr a ->+ Expr Bool+(.<=.) = lift2Decorated (<=) "leq" (Just ".<=.") False 4++(.>=.) ::+ (Columnable a, Ord a) =>+ Expr a ->+ Expr a ->+ Expr Bool+(.>=.) = lift2Decorated (>=) "geq" (Just ".>=.") False 4++(.+.) :: (Columnable a, Num a) => Expr a -> Expr a -> Expr a+(.+.) = (+)++(.-.) :: (Columnable a, Num a) => Expr a -> Expr a -> Expr a+(.-.) = (-)++(.*.) :: (Columnable a, Num a) => Expr a -> Expr a -> Expr a+(.*.) = (*)++(./.) :: (Columnable a, Fractional a) => Expr a -> Expr a -> Expr a+(./.) = (/)++-- Nullable-aware arithmetic operators++{- | Nullable-aware addition. Works for all combinations of nullable\/non-nullable operands.+@col \@Int "x" .+ col \@(Maybe Int) "y" -- :: Expr (Maybe Int)@+-}+(.+) ::+ ( NumericWidenOp (BaseType a) (BaseType b)+ , NullLift2Op a b (Promote (BaseType a) (BaseType b)) (WidenResult a b)+ , Num (Promote (BaseType a) (BaseType b))+ ) =>+ Expr a ->+ Expr b ->+ Expr (WidenResult a b)+(.+) = lift2Decorated (applyNull2 (widenArithOp (+))) "nulladd" (Just ".+") True 6++-- | Nullable-aware subtraction.+(.-) ::+ ( NumericWidenOp (BaseType a) (BaseType b)+ , NullLift2Op a b (Promote (BaseType a) (BaseType b)) (WidenResult a b)+ , Num (Promote (BaseType a) (BaseType b))+ ) =>+ Expr a ->+ Expr b ->+ Expr (WidenResult a b)+(.-) = lift2Decorated (applyNull2 (widenArithOp (-))) "nullsub" (Just ".-") False 6++-- | Nullable-aware multiplication.+(.*) ::+ ( NumericWidenOp (BaseType a) (BaseType b)+ , NullLift2Op a b (Promote (BaseType a) (BaseType b)) (WidenResult a b)+ , Num (Promote (BaseType a) (BaseType b))+ ) =>+ Expr a ->+ Expr b ->+ Expr (WidenResult a b)+(.*) = lift2Decorated (applyNull2 (widenArithOp (*))) "nullmul" (Just ".*") True 7++-- | Nullable-aware division. Integral operands are promoted to Double.+(./) ::+ ( DivWidenOp (BaseType a) (BaseType b)+ , NullLift2Op a b (PromoteDiv (BaseType a) (BaseType b)) (WidenResultDiv a b)+ , Fractional (PromoteDiv (BaseType a) (BaseType b))+ ) =>+ Expr a ->+ Expr b ->+ Expr (WidenResultDiv a b)+(./) = lift2Decorated (applyNull2 (divArithOp (/))) "nulldiv" (Just "./") False 7++-- Nullable-aware comparison operators (three-valued logic: Nothing if either operand is Nothing)++{- | Nullable-aware equality. Widens numeric operands to their common type,+so @Expr Double .== Expr Int@ typechecks. Returns @Maybe Bool@ when either+operand is nullable.+-}+(.==) ::+ ( NumericWidenOp (BaseType a) (BaseType b)+ , NullLift2Op a b Bool (NullCmpResult a b)+ , Eq (Promote (BaseType a) (BaseType b))+ ) =>+ Expr a ->+ Expr b ->+ Expr (NullCmpResult a b)+(.==) = Binary NullEq++-- | Nullable-aware inequality. Widens numeric operands to their common type.+(./=) ::+ ( NumericWidenOp (BaseType a) (BaseType b)+ , NullLift2Op a b Bool (NullCmpResult a b)+ , Eq (Promote (BaseType a) (BaseType b))+ ) =>+ Expr a ->+ Expr b ->+ Expr (NullCmpResult a b)+(./=) = Binary NullNeq++-- | Nullable-aware less-than. Widens numeric operands to their common type.+(.<) ::+ ( NumericWidenOp (BaseType a) (BaseType b)+ , NullLift2Op a b Bool (NullCmpResult a b)+ , Ord (Promote (BaseType a) (BaseType b))+ ) =>+ Expr a ->+ Expr b ->+ Expr (NullCmpResult a b)+(.<) = Binary NullLt++-- | Nullable-aware greater-than. Widens numeric operands to their common type.+(.>) ::+ ( NumericWidenOp (BaseType a) (BaseType b)+ , NullLift2Op a b Bool (NullCmpResult a b)+ , Ord (Promote (BaseType a) (BaseType b))+ ) =>+ Expr a ->+ Expr b ->+ Expr (NullCmpResult a b)+(.>) = Binary NullGt++{- | Nullable-aware less-than-or-equal. Widens numeric operands to their+common type, so @Expr Double .<= Expr Int@ typechecks.+-}+(.<=) ::+ ( NumericWidenOp (BaseType a) (BaseType b)+ , NullLift2Op a b Bool (NullCmpResult a b)+ , Ord (Promote (BaseType a) (BaseType b))+ ) =>+ Expr a ->+ Expr b ->+ Expr (NullCmpResult a b)+(.<=) = Binary NullLeq++-- | Nullable-aware greater-than-or-equal. Widens numeric operands to their common type.+(.>=) ::+ ( NumericWidenOp (BaseType a) (BaseType b)+ , NullLift2Op a b Bool (NullCmpResult a b)+ , Ord (Promote (BaseType a) (BaseType b))+ ) =>+ Expr a ->+ Expr b ->+ Expr (NullCmpResult a b)+(.>=) = Binary NullGeq++(.&&.) :: Expr Bool -> Expr Bool -> Expr Bool+(.&&.) = lift2Decorated (&&) "and" (Just ".&&.") True 3++(.||.) :: Expr Bool -> Expr Bool -> Expr Bool+(.||.) = lift2Decorated (||) "or" (Just ".||.") True 2++-- | Nullable-aware logical AND. Returns @Maybe Bool@ when either operand is nullable.+(.&&) ::+ (NullableCmpOp a b (NullCmpResult a b), BaseType a ~ Bool) =>+ Expr a ->+ Expr b ->+ Expr (NullCmpResult a b)+(.&&) = Binary NullAnd++-- | Nullable-aware logical OR. Returns @Maybe Bool@ when either operand is nullable.+(.||) ::+ (NullableCmpOp a b (NullCmpResult a b), BaseType a ~ Bool) =>+ Expr a ->+ Expr b ->+ Expr (NullCmpResult a b)+(.||) = Binary NullOr++(.^^) ::+ ( Columnable (BaseType a)+ , Columnable (BaseType b)+ , Fractional (BaseType a)+ , Integral (BaseType b)+ , NumericWidenOp (BaseType a) (BaseType b)+ , NullLift2Op a b (BaseType a) a+ , Num (Promote (BaseType a) (BaseType b))+ ) =>+ Expr a -> Expr b -> Expr a+(.^^) = lift2Decorated (applyNull2 (^^)) "pow" (Just ".^^") False 8++(.^) ::+ ( Columnable (BaseType a)+ , Columnable (BaseType b)+ , Num (BaseType a)+ , Integral (BaseType b)+ , NumericWidenOp (BaseType a) (BaseType b)+ , NullLift2Op a b (BaseType a) a+ , Num (Promote (BaseType a) (BaseType b))+ ) =>+ Expr a -> Expr b -> Expr a+(.^) = lift2Decorated (applyNull2 (^)) "pow" (Just ".^") False 8++-- Same-type (non-nullable) exponentiation operators++(.^^.) ::+ (Columnable a, Columnable b, Fractional a, Integral b) =>+ Expr a -> Expr b -> Expr a+(.^^.) = lift2Decorated (^^) "pow" (Just ".^^.") False 8++(.^.) ::+ (Columnable a, Columnable b, Num a, Integral b) =>+ Expr a -> Expr b -> Expr a+(.^.) = lift2Decorated (^) "pow" (Just ".^.") False 8
+ src-internal/DataFrame/Internal/Expression/Operators/Nullable.hs view
@@ -0,0 +1,467 @@+{-# LANGUAGE AllowAmbiguousTypes #-}+{-# LANGUAGE FlexibleContexts #-}+{-# LANGUAGE FlexibleInstances #-}+{-# LANGUAGE FunctionalDependencies #-}+{-# LANGUAGE ScopedTypeVariables #-}+{-# LANGUAGE TypeApplications #-}+{-# LANGUAGE TypeFamilies #-}+{-# LANGUAGE TypeOperators #-}+{-# LANGUAGE UndecidableInstances #-}+{-# LANGUAGE UndecidableSuperClasses #-}++{- | Nullable-aware arithmetic and comparison operators ('.+', '.==', …) that work+transparently across nullable (@Maybe a@) and non-nullable (@a@) operands.+Functional dependencies infer the result type without annotations.++@+-- Mixing nullable and non-nullable columns:+F.col \@Int \"x\" '.+' F.col \@(Maybe Int) \"y\" -- :: Expr (Maybe Int)++-- Both non-nullable (existing behaviour preserved):+F.col \@Int \"x\" '.+' F.col \@Int \"y\" -- :: Expr Int++-- Comparison with three-valued logic:+F.col \@(Maybe Int) \"x\" '.==' F.col \@Int \"y\" -- :: Expr (Maybe Bool)+@+-}+module DataFrame.Internal.Expression.Operators.Nullable (+ -- * Type family+ BaseType,++ -- * Arithmetic class+ NullableArithOp (..),++ -- * Comparison class+ NullableCmpOp (..),++ -- * Generalized nullable lift classes+ NullLift1Op (..),+ NullLift2Op (..),++ -- * Result-type type families (drive inference in nullLift / nullLift2)+ NullLift1Result,+ NullLift2Result,++ -- * Result-type type family for comparison operators+ NullCmpResult,++ -- * Numeric widening+ NumericWidenOp (..),+ widenArithOp,+ widenCmpOp,+ WidenResult,++ -- * Division widening (integral × integral → Double)+ DivWidenOp (..),+ divArithOp,+ WidenResultDiv,+) where++import Data.Int (Int32, Int64)+import DataFrame.Internal.Column (Columnable)+import DataFrame.Internal.Column.Types (Promote, PromoteDiv)++{- | Strip one layer of 'Maybe'.++@+BaseType (Maybe a) = a+BaseType a = a -- for any non-Maybe type+@+-}+type family BaseType a where+ BaseType (Maybe a) = a+ BaseType a = a++{- | Arithmetic binary operations that work over nullable and non-nullable operand+types. The functional dependency @a b -> c@ infers the result; the 'OVERLAPPABLE'+non-nullable instance yields to the specific @(Maybe a, Maybe a)@ one.+-}+class+ ( Columnable a+ , Columnable b+ , Columnable c+ ) =>+ NullableArithOp a b c+ | a b -> c+ where+ {- | Lift an arithmetic function over the inner (non-Maybe) values.+ 'Nothing' short-circuits: any 'Nothing' operand produces 'Nothing'.+ -}+ nullArithOp ::+ (BaseType a -> BaseType a -> BaseType a) ->+ a ->+ b ->+ c++{- | Compute the result type of a nullable comparison.++@+NullCmpResult (Maybe a) b = Maybe Bool+NullCmpResult a (Maybe b) = Maybe Bool -- when a is apart from Maybe+NullCmpResult a b = Bool+@++Used by the comparison operators ('.==', '.<', etc.) so GHC infers the+return type without an explicit annotation.+-}+type family NullCmpResult a b where+ NullCmpResult (Maybe a) b = Maybe Bool+ NullCmpResult a (Maybe b) = Maybe Bool+ NullCmpResult a b = Bool++{- | Comparison binary operations over nullable and non-nullable operands. No+functional dependency on @e@; overlapping/overlappable instance pragmas pick the+unique most-specific instance from the concrete operand types.+-}+class+ ( Columnable a+ , Columnable b+ , Columnable e+ ) =>+ NullableCmpOp a b e+ where+ {- | Lift a comparison function over the inner values (three-valued logic).+ Returns 'Nothing' when either operand is 'Nothing'.+ -}+ nullCmpOp ::+ (BaseType a -> BaseType a -> Bool) ->+ a ->+ b ->+ e++{- | Non-nullable × Non-nullable: apply directly, no wrapping.+Arithmetic result is @a@; comparison result is @Bool@.+-}+instance+ {-# OVERLAPPABLE #-}+ (Columnable a, a ~ BaseType a) =>+ NullableArithOp a a a+ where+ nullArithOp f = f++instance+ {-# OVERLAPPABLE #-}+ (Columnable a, Columnable Bool, a ~ BaseType a) =>+ NullableCmpOp a a Bool+ where+ nullCmpOp f = f++-- | Nullable × Non-nullable: 'Nothing' short-circuits.+instance+ (Columnable a, Columnable (Maybe a)) =>+ NullableArithOp (Maybe a) a (Maybe a)+ where+ nullArithOp _f Nothing _ = Nothing+ nullArithOp f (Just x) y = Just (f x y)++instance+ (Columnable a, Columnable (Maybe a), Columnable (Maybe Bool)) =>+ NullableCmpOp (Maybe a) a (Maybe Bool)+ where+ nullCmpOp _f Nothing _ = Nothing+ nullCmpOp f (Just x) y = Just (f x y)++-- | Non-nullable × Nullable: 'Nothing' short-circuits.+instance+ ( Columnable a+ , Columnable (Maybe a)+ , a ~ BaseType a+ ) =>+ NullableArithOp a (Maybe a) (Maybe a)+ where+ nullArithOp _f _ Nothing = Nothing+ nullArithOp f x (Just y) = Just (f x y)++instance+ ( Columnable a+ , Columnable (Maybe a)+ , Columnable (Maybe Bool)+ , a ~ BaseType a+ ) =>+ NullableCmpOp a (Maybe a) (Maybe Bool)+ where+ nullCmpOp _f _ Nothing = Nothing+ nullCmpOp f x (Just y) = Just (f x y)++-- | Nullable × Nullable: either 'Nothing' short-circuits.+instance+ {-# OVERLAPPING #-}+ (Columnable a, Columnable (Maybe a)) =>+ NullableArithOp (Maybe a) (Maybe a) (Maybe a)+ where+ nullArithOp _f Nothing _ = Nothing+ nullArithOp _f _ Nothing = Nothing+ nullArithOp f (Just x) (Just y) = Just (f x y)++instance+ {-# OVERLAPPING #-}+ (Columnable a, Columnable (Maybe a), Columnable (Maybe Bool)) =>+ NullableCmpOp (Maybe a) (Maybe a) (Maybe Bool)+ where+ nullCmpOp _f Nothing _ = Nothing+ nullCmpOp _f _ Nothing = Nothing+ nullCmpOp f (Just x) (Just y) = Just (f x y)++-- ---------------------------------------------------------------------------+-- Generalized nullable lift (unary)+-- ---------------------------------------------------------------------------++{- | Lift a unary function over a column expression, propagating 'Nothing' (applied+directly when non-nullable, under 'Just' when @a = Maybe x@). Use via+'DataFrame.Functions.nullLift'.+-}++{- | Compute the result type of a nullable unary lift.++@+NullLift1Result (Maybe a) r = Maybe r+NullLift1Result a r = r -- for any non-Maybe a+@++Used by 'DataFrame.Functions.nullLift' so GHC can infer the return type+without an explicit annotation.+-}+type family NullLift1Result a r where+ NullLift1Result (Maybe a) r = Maybe r+ NullLift1Result a r = r++class+ ( Columnable a+ , Columnable r+ , Columnable c+ ) =>+ NullLift1Op a r c+ where+ applyNull1 :: (BaseType a -> r) -> a -> c++-- | Non-nullable: apply directly.+instance+ {-# OVERLAPPABLE #-}+ (Columnable a, Columnable r, a ~ BaseType a) =>+ NullLift1Op a r r+ where+ applyNull1 f = f++-- | Nullable: propagate 'Nothing'.+instance+ {-# OVERLAPPING #-}+ (Columnable a, Columnable r, Columnable (Maybe r)) =>+ NullLift1Op (Maybe a) r (Maybe r)+ where+ applyNull1 _ Nothing = Nothing+ applyNull1 f (Just x) = Just (f x)++-- ---------------------------------------------------------------------------+-- Generalized nullable lift (binary)+-- ---------------------------------------------------------------------------++{- | Lift a binary function over two column expressions, propagating 'Nothing': the+result is @Maybe r@ if either operand is nullable, else @r@. Use via+'DataFrame.Functions.nullLift2'.+-}++{- | Compute the result type of a nullable binary lift.++@+NullLift2Result (Maybe a) b r = Maybe r+NullLift2Result a (Maybe b) r = Maybe r -- when a is apart from Maybe+NullLift2Result a b r = r+@++Used by 'DataFrame.Functions.nullLift2' so GHC can infer the return type.+-}+type family NullLift2Result a b r where+ NullLift2Result (Maybe a) b r = Maybe r+ NullLift2Result a (Maybe b) r = Maybe r+ NullLift2Result a b r = r++class+ ( Columnable a+ , Columnable b+ , Columnable r+ , Columnable c+ ) =>+ NullLift2Op a b r c+ where+ applyNull2 :: (BaseType a -> BaseType b -> r) -> a -> b -> c++-- | Both non-nullable: apply directly.+instance+ {-# OVERLAPPABLE #-}+ (Columnable a, Columnable b, Columnable r, a ~ BaseType a, b ~ BaseType b) =>+ NullLift2Op a b r r+ where+ applyNull2 f = f++-- | Left nullable: 'Nothing' short-circuits.+instance+ {-# OVERLAPPABLE #-}+ (Columnable a, Columnable b, Columnable r, Columnable (Maybe r), b ~ BaseType b) =>+ NullLift2Op (Maybe a) b r (Maybe r)+ where+ applyNull2 _ Nothing _ = Nothing+ applyNull2 f (Just x) y = Just (f x y)++-- | Right nullable: 'Nothing' short-circuits.+instance+ {-# OVERLAPPABLE #-}+ (Columnable a, Columnable b, Columnable r, Columnable (Maybe r), a ~ BaseType a) =>+ NullLift2Op a (Maybe b) r (Maybe r)+ where+ applyNull2 _ _ Nothing = Nothing+ applyNull2 f x (Just y) = Just (f x y)++-- | Both nullable: either 'Nothing' short-circuits.+instance+ {-# OVERLAPPING #-}+ (Columnable a, Columnable b, Columnable r, Columnable (Maybe r)) =>+ NullLift2Op (Maybe a) (Maybe b) r (Maybe r)+ where+ applyNull2 _ Nothing _ = Nothing+ applyNull2 _ _ Nothing = Nothing+ applyNull2 f (Just x) (Just y) = Just (f x y)++-- ---------------------------------------------------------------------------+-- Numeric widening+-- ---------------------------------------------------------------------------++{- | Widen two numeric base types to their promoted common type.++When @a ~ b@ the coercions are identity; otherwise one operand is widened+(e.g. 'Int' → 'Double').+-}+class (Columnable (Promote a b)) => NumericWidenOp a b where+ widen1 :: a -> Promote a b+ widen2 :: b -> Promote a b++-- | Same type: identity coercions.+instance {-# OVERLAPPING #-} (Columnable a) => NumericWidenOp a a where+ widen1 = id+ widen2 = id++instance NumericWidenOp Int Double where widen1 = fromIntegral; widen2 = id+instance NumericWidenOp Double Int where+ widen1 = id+ widen2 = fromIntegral+instance NumericWidenOp Float Double where widen1 = realToFrac; widen2 = id+instance NumericWidenOp Double Float where+ widen1 = id+ widen2 = realToFrac+instance NumericWidenOp Int32 Float where widen1 = fromIntegral; widen2 = id+instance NumericWidenOp Float Int32 where+ widen1 = id+ widen2 = fromIntegral+instance NumericWidenOp Int32 Double where widen1 = fromIntegral; widen2 = id+instance NumericWidenOp Double Int32 where+ widen1 = id+ widen2 = fromIntegral+instance NumericWidenOp Int64 Float where widen1 = fromIntegral; widen2 = id+instance NumericWidenOp Float Int64 where+ widen1 = id+ widen2 = fromIntegral+instance NumericWidenOp Int64 Double where widen1 = fromIntegral; widen2 = id+instance NumericWidenOp Double Int64 where+ widen1 = id+ widen2 = fromIntegral++-- | Apply an arithmetic function after widening both operands to their common type.+widenArithOp ::+ forall a b.+ (NumericWidenOp a b) =>+ (Promote a b -> Promote a b -> Promote a b) ->+ a ->+ b ->+ Promote a b+widenArithOp f x y = f (widen1 @a @b x) (widen2 @a @b y)++-- | Apply a comparison function after widening both operands to their common type.+widenCmpOp ::+ forall a b.+ (NumericWidenOp a b) =>+ (Promote a b -> Promote a b -> Bool) ->+ a ->+ b ->+ Bool+widenCmpOp f x y = f (widen1 @a @b x) (widen2 @a @b y)++-- | Result type of a widening binary operator, accounting for nullable wrappers.+type WidenResult a b = NullLift2Result a b (Promote (BaseType a) (BaseType b))++-- ---------------------------------------------------------------------------+-- Division widening (integral × integral → Double)+-- ---------------------------------------------------------------------------++{- | Like 'NumericWidenOp' but uses 'PromoteDiv': integral×integral → Double.+Floating types still dominate (Double > Float), and any two integral types+(same or mixed) are both widened to Double.+-}+class (Columnable (PromoteDiv a b)) => DivWidenOp a b where+ divWiden1 :: a -> PromoteDiv a b+ divWiden2 :: b -> PromoteDiv a b++-- Floating same-type (identity)+instance DivWidenOp Double Double where divWiden1 = id; divWiden2 = id+instance DivWidenOp Float Float where divWiden1 = id; divWiden2 = id++-- Mixed Double/Float+instance DivWidenOp Double Float where divWiden1 = id; divWiden2 = realToFrac+instance DivWidenOp Float Double where divWiden1 = realToFrac; divWiden2 = id++-- Double beats integral+instance DivWidenOp Double Int where divWiden1 = id; divWiden2 = fromIntegral+instance DivWidenOp Int Double where divWiden1 = fromIntegral; divWiden2 = id+instance DivWidenOp Double Int32 where divWiden1 = id; divWiden2 = fromIntegral+instance DivWidenOp Int32 Double where divWiden1 = fromIntegral; divWiden2 = id+instance DivWidenOp Double Int64 where divWiden1 = id; divWiden2 = fromIntegral+instance DivWidenOp Int64 Double where divWiden1 = fromIntegral; divWiden2 = id++-- Float beats integral+instance DivWidenOp Float Int where divWiden1 = id; divWiden2 = fromIntegral+instance DivWidenOp Int Float where divWiden1 = fromIntegral; divWiden2 = id+instance DivWidenOp Float Int32 where divWiden1 = id; divWiden2 = fromIntegral+instance DivWidenOp Int32 Float where divWiden1 = fromIntegral; divWiden2 = id+instance DivWidenOp Float Int64 where divWiden1 = id; divWiden2 = fromIntegral+instance DivWidenOp Int64 Float where divWiden1 = fromIntegral; divWiden2 = id++-- Integral × integral → Double+instance DivWidenOp Int Int where+ divWiden1 = fromIntegral+ divWiden2 = fromIntegral+instance DivWidenOp Int32 Int32 where+ divWiden1 = fromIntegral+ divWiden2 = fromIntegral+instance DivWidenOp Int64 Int64 where+ divWiden1 = fromIntegral+ divWiden2 = fromIntegral+instance DivWidenOp Int Int32 where+ divWiden1 = fromIntegral+ divWiden2 = fromIntegral+instance DivWidenOp Int32 Int where+ divWiden1 = fromIntegral+ divWiden2 = fromIntegral+instance DivWidenOp Int Int64 where+ divWiden1 = fromIntegral+ divWiden2 = fromIntegral+instance DivWidenOp Int64 Int where+ divWiden1 = fromIntegral+ divWiden2 = fromIntegral+instance DivWidenOp Int32 Int64 where+ divWiden1 = fromIntegral+ divWiden2 = fromIntegral+instance DivWidenOp Int64 Int32 where+ divWiden1 = fromIntegral+ divWiden2 = fromIntegral++-- | Apply an arithmetic function after widening both operands via 'PromoteDiv'.+divArithOp ::+ forall a b.+ (DivWidenOp a b) =>+ (PromoteDiv a b -> PromoteDiv a b -> PromoteDiv a b) ->+ a ->+ b ->+ PromoteDiv a b+divArithOp f x y = f (divWiden1 @a @b x) (divWiden2 @a @b y)++-- | Result type of a division-widening binary operator, accounting for nullable wrappers.+type WidenResultDiv a b =+ NullLift2Result a b (PromoteDiv (BaseType a) (BaseType b))
+ src-internal/DataFrame/Internal/Expression/Simplify.hs view
@@ -0,0 +1,417 @@+{-# LANGUAGE AllowAmbiguousTypes #-}+{-# LANGUAGE FlexibleContexts #-}+{-# LANGUAGE GADTs #-}+{-# LANGUAGE MultiWayIf #-}+{-# LANGUAGE OverloadedStrings #-}+{-# LANGUAGE PatternSynonyms #-}+{-# LANGUAGE ScopedTypeVariables #-}+{-# LANGUAGE TypeApplications #-}++module DataFrame.Internal.Expression.Simplify (+ simplify,+ simplifyPredicatePair,++ -- * Path-condition entailment (for fitted-tree pruning)+ PredFact,+ factTrue,+ factFalse,+ entails,+) where++import Control.Monad (guard)+import Data.Maybe (fromMaybe)+import Data.Type.Equality (testEquality, (:~:) (Refl))+import Type.Reflection (eqTypeRep, typeRep, (:~~:) (HRefl), pattern App)++import DataFrame.Internal.Column (Columnable)+import DataFrame.Internal.Expression (+ BinaryOp,+ Expr (..),+ UnaryOp (unaryName),+ eqExpr,+ normalize,+ )+import DataFrame.Internal.Expression.Operators (+ NullAnd,+ NullEq,+ NullGeq,+ NullGt,+ NullLeq,+ NullLt,+ NullNeq,+ NullOr,+ (.==.),+ )++simplify :: forall a. (Columnable a) => Expr a -> Expr a+simplify e+ | isBoolish @a = fixpoint (10 :: Int) e+ | otherwise = e+ where+ fixpoint 0 x = x+ fixpoint n x = let x' = simplifyB x in if eqExpr x x' then x else fixpoint (n - 1) x'++isBoolish :: forall a. (Columnable a) => Bool+isBoolish =+ case ( testEquality (typeRep @a) (typeRep @Bool)+ , testEquality (typeRep @a) (typeRep @(Maybe Bool))+ ) of+ (Just Refl, _) -> True+ (_, Just Refl) -> True+ _ -> False++data Conn = ConnAnd | ConnOr++connOf :: forall op c b r. (BinaryOp op) => op c b r -> Maybe Conn+connOf _+ | Just HRefl <- eqTypeRep (typeRep @op) (typeRep @NullAnd) = Just ConnAnd+ | Just HRefl <- eqTypeRep (typeRep @op) (typeRep @NullOr) = Just ConnOr+ | otherwise = Nothing++simplifyB :: forall a. (Columnable a) => Expr a -> Expr a+simplifyB expr = case expr of+ Binary (op :: op c b a) l r+ | Just conn <- connOf op+ , Just Refl <- testEquality (typeRep @c) (typeRep @a)+ , Just Refl <- testEquality (typeRep @b) (typeRep @a) ->+ let l' = simplifyB l; r' = simplifyB r+ in fromMaybe (Binary op l' r') (combine conn l' r')+ | otherwise -> expr+ Unary (op :: op b a) inner+ | Just Refl <- testEquality (typeRep @a) (typeRep @Bool)+ , Just Refl <- testEquality (typeRep @b) (typeRep @Bool)+ , unaryName op == "not" ->+ simplifyNot op (simplifyB inner)+ | otherwise -> expr+ If c t f ->+ let c' = simplify c+ t' = simplifyB t+ f' = simplifyB f+ in case asBoolLit c' of+ Just True -> t'+ Just False -> f'+ Nothing+ | eqExpr t' f' -> t'+ | Just Refl <- testEquality (typeRep @a) (typeRep @Bool)+ , asBoolLit t' == Just True+ , asBoolLit f' == Just False ->+ c'+ | otherwise -> If c' t' f'+ _ -> expr++simplifyNot :: (UnaryOp op) => op Bool Bool -> Expr Bool -> Expr Bool+simplifyNot op inner = case asBoolLit inner of+ Just b -> Lit (not b)+ Nothing -> case inner of+ Unary (op2 :: op2 b2 Bool) inner2+ | unaryName op2 == "not"+ , Just Refl <- testEquality (typeRep @b2) (typeRep @Bool) ->+ inner2+ _ -> Unary op inner++combine :: (Columnable a) => Conn -> Expr a -> Expr a -> Maybe (Expr a)+combine ConnAnd = combineAnd+combine ConnOr = combineOr++asBoolLit :: forall a. (Columnable a) => Expr a -> Maybe Bool+asBoolLit (Lit v) =+ case testEquality (typeRep @a) (typeRep @Bool) of+ Just Refl -> Just v+ Nothing -> case testEquality (typeRep @a) (typeRep @(Maybe Bool)) of+ Just Refl -> v+ Nothing -> Nothing+asBoolLit _ = Nothing++{- | Polymorphic boolean literal: @Lit b@ for @Expr Bool@, @Lit (Just b)@ for+@Expr (Maybe Bool)@.+-}+litBoolish :: forall a. (Columnable a) => Bool -> Maybe (Expr a)+litBoolish v =+ case testEquality (typeRep @a) (typeRep @Bool) of+ Just Refl -> Just (Lit v)+ Nothing -> case testEquality (typeRep @a) (typeRep @(Maybe Bool)) of+ Just Refl -> Just (Lit (Just v))+ Nothing -> Nothing++combineAnd :: (Columnable a) => Expr a -> Expr a -> Maybe (Expr a)+combineAnd l r+ | eqExpr l r = Just l+ | asBoolLit l == Just False = litBoolish False+ | asBoolLit r == Just False = litBoolish False+ | asBoolLit l == Just True = Just r+ | asBoolLit r == Just True = Just l+ | absorbs ConnOr l r = Just l+ | absorbs ConnOr r l = Just r+ | otherwise = simplifyPredicatePair True l r++combineOr :: (Columnable a) => Expr a -> Expr a -> Maybe (Expr a)+combineOr l r+ | eqExpr l r = Just l+ | asBoolLit l == Just True = litBoolish True+ | asBoolLit r == Just True = litBoolish True+ | asBoolLit l == Just False = Just r+ | asBoolLit r == Just False = Just l+ | absorbs ConnAnd l r = Just l+ | absorbs ConnAnd r l = Just r+ | otherwise = simplifyPredicatePair False l r++absorbs :: (Columnable a) => Conn -> Expr a -> Expr a -> Bool+absorbs conn x (Binary (op :: op c b a) ya yb)+ | Just c' <- connOf op+ , sameConn conn c'+ , Just Refl <- testEquality (typeRep @c) (typeRep @a)+ , Just Refl <- testEquality (typeRep @b) (typeRep @a) =+ eqExpr x ya || eqExpr x yb+absorbs _ _ _ = False++sameConn :: Conn -> Conn -> Bool+sameConn ConnAnd ConnAnd = True+sameConn ConnOr ConnOr = True+sameConn _ _ = False++data Cmp = CLt | CLeq | CGt | CGeq | CEq | CNeq deriving (Eq)++data NullK = Total | FalseOnNull | UnknownOnNull deriving (Eq)++data Atom = Atom+ { aCmp :: Cmp+ , aThr :: !Double+ , aKey :: String+ , aNull :: NullK+ , aIntegral :: Bool+ }++cmpOf :: forall op c b r. (BinaryOp op) => op c b r -> Maybe Cmp+cmpOf _+ | Just HRefl <- eqTypeRep (typeRep @op) (typeRep @NullLt) = Just CLt+ | Just HRefl <- eqTypeRep (typeRep @op) (typeRep @NullLeq) = Just CLeq+ | Just HRefl <- eqTypeRep (typeRep @op) (typeRep @NullGt) = Just CGt+ | Just HRefl <- eqTypeRep (typeRep @op) (typeRep @NullGeq) = Just CGeq+ | Just HRefl <- eqTypeRep (typeRep @op) (typeRep @NullEq) = Just CEq+ | Just HRefl <- eqTypeRep (typeRep @op) (typeRep @NullNeq) = Just CNeq+ | otherwise = Nothing++isLower, isUpper :: Cmp -> Bool+isLower c = c == CGt || c == CGeq+isUpper c = c == CLt || c == CLeq++-- | True if @x@ is a @Maybe _@ type.+isMaybeTy :: forall x. (Columnable x) => Bool+isMaybeTy = case typeRep @x of+ App con _ -> case eqTypeRep con (typeRep @Maybe) of Just HRefl -> True; _ -> False+ _ -> False++litDouble :: forall b. (Columnable b) => Expr b -> Maybe Double+litDouble (Lit v) =+ case testEquality (typeRep @b) (typeRep @Double) of+ Just Refl -> Just v+ Nothing -> case testEquality (typeRep @b) (typeRep @Int) of+ Just Refl -> Just (fromIntegral v)+ Nothing -> case testEquality (typeRep @b) (typeRep @(Maybe Double)) of+ Just Refl -> v+ Nothing -> case testEquality (typeRep @b) (typeRep @(Maybe Int)) of+ Just Refl -> fromIntegral <$> v+ Nothing -> Nothing+litDouble _ = Nothing++{- | True for a column lifted from an integral type (never NaN): @toDouble (col …)@+or a column whose type is itself integral.+-}+integralColE :: forall c. (Columnable c) => Expr c -> Bool+integralColE (Unary op _) = unaryName op == "toDouble"+integralColE _ =+ or+ [ matches @Int+ , matches @(Maybe Int)+ ]+ where+ matches :: forall t. (Columnable t) => Bool+ matches = case testEquality (typeRep @c) (typeRep @t) of Just Refl -> True; _ -> False++atomOf :: forall a. (Columnable a) => Expr a -> Maybe Atom+atomOf (Unary fm (Binary (op :: op c b r) (colE :: Expr c) litE))+ | unaryName fm == "fromMaybe"+ , Just cmp <- cmpOf op+ , Just t <- litDouble litE =+ Just (Atom cmp t (show (normalize colE)) FalseOnNull (integralColE colE))+atomOf (Binary (op :: op c b a) (colE :: Expr c) litE)+ | Just cmp <- cmpOf op+ , Just t <- litDouble litE =+ let nk = if isMaybeTy @c then UnknownOnNull else Total+ in Just (Atom cmp t (show (normalize colE)) nk (integralColE colE))+atomOf _ = Nothing++simplifyPredicatePair ::+ forall a. (Columnable a) => Bool -> Expr a -> Expr a -> Maybe (Expr a)+simplifyPredicatePair isAnd a b = do+ atomA <- atomOf a+ atomB <- atomOf b+ guard (aKey atomA == aKey atomB)+ let nk = aNull atomA+ integral = aIntegral atomA+ if isAnd+ then andAtoms a atomA b atomB nk integral+ else orAtoms a atomA b atomB nk integral++-- | Contradiction folds to a literal False unless null-rows make it unknown.+litFalseGated :: (Columnable a) => NullK -> Maybe (Expr a)+litFalseGated UnknownOnNull = Nothing+litFalseGated _ = litBoolish False++{- | Tautology to literal True is sound only for total (never-null) atoms; the+exhaustive-cover form additionally needs a non-NaN (integral) column.+-}+litTrueTotal :: (Columnable a) => NullK -> Maybe (Expr a)+litTrueTotal Total = litBoolish True+litTrueTotal _ = Nothing++andAtoms ::+ (Columnable a) =>+ Expr a -> Atom -> Expr a -> Atom -> NullK -> Bool -> Maybe (Expr a)+andAtoms a atomA b atomB nk _ =+ let cA = aCmp atomA; tA = aThr atomA; cB = aCmp atomB; tB = aThr atomB+ in if+ | isLower cA, isLower cB, cA == cB -> Just (if tA >= tB then a else b)+ | isUpper cA, isUpper cB, cA == cB -> Just (if tA <= tB then a else b)+ | isLower cA, isUpper cB -> lu cA tA cB tB+ | isUpper cA, isLower cB -> lu cB tB cA tA+ | cA == CEq, cB == CEq -> if tA == tB then Just a else litFalseGated nk+ | cA == CEq, cB == CNeq -> if tA == tB then litFalseGated nk else Just a+ | cA == CNeq, cB == CEq -> if tA == tB then litFalseGated nk else Just b+ | cA == CEq -> if satisfies tA cB tB then Just a else litFalseGated nk+ | cB == CEq -> if satisfies tB cA tA then Just b else litFalseGated nk+ | cA == CNeq, cB == CNeq -> Nothing+ | cA == CNeq -> if outside tA cB tB then Just b else Nothing+ | cB == CNeq -> if outside tB cA tA then Just a else Nothing+ | otherwise -> Nothing+ where+ lu lc lo uc hi+ | lo > hi = litFalseGated nk+ | lo == hi, lc == CGeq, uc == CLeq = pointEq a lo+ | lo == hi = litFalseGated nk+ | otherwise = Nothing++orAtoms ::+ (Columnable a) =>+ Expr a -> Atom -> Expr a -> Atom -> NullK -> Bool -> Maybe (Expr a)+orAtoms a atomA b atomB nk integral =+ let cA = aCmp atomA; tA = aThr atomA; cB = aCmp atomB; tB = aThr atomB+ in if+ | isLower cA, isLower cB, cA == cB -> Just (if tA <= tB then a else b)+ | isUpper cA, isUpper cB, cA == cB -> Just (if tA >= tB then a else b)+ | isUpper cA+ , isLower cB+ , nk == Total+ , integral+ , covers cB tB cA tA ->+ litTrueTotal nk+ | isLower cA+ , isUpper cB+ , nk == Total+ , integral+ , covers cA tA cB tB ->+ litTrueTotal nk+ | cA == CNeq, cB == CNeq -> if tA == tB then Just a else litTrueTotal nk+ | cA == CEq, cB == CNeq -> if tA == tB then litTrueTotal nk else Just b+ | cA == CNeq, cB == CEq -> if tA == tB then litTrueTotal nk else Just a+ | cA == CEq, cB == CEq -> if tA == tB then Just a else Nothing+ | otherwise -> Nothing++{- | Build @col == t@ for the point-collapse rule; only strict @Expr Bool@ over a+@Double@ column (otherwise bail).+-}+pointEq :: forall a. (Columnable a) => Expr a -> Double -> Maybe (Expr a)+pointEq atom lo = case testEquality (typeRep @a) (typeRep @Bool) of+ Just Refl -> (\colE -> colE .==. Lit lo) <$> recoverColD atom+ Nothing -> Nothing++recoverColD :: Expr x -> Maybe (Expr Double)+recoverColD (Binary _ (colE :: Expr c) _) =+ case testEquality (typeRep @c) (typeRep @Double) of+ Just Refl -> Just colE+ _ -> Nothing+recoverColD (Unary _ inner) = recoverColD inner+recoverColD _ = Nothing++covers :: Cmp -> Double -> Cmp -> Double -> Bool+covers lowerCmp lo upperCmp hi =+ lo < hi || (lo == hi && (lowerCmp == CGeq || upperCmp == CLeq))++satisfies :: Double -> Cmp -> Double -> Bool+satisfies t CGt tb = t > tb+satisfies t CGeq tb = t >= tb+satisfies t CLt tb = t < tb+satisfies t CLeq tb = t <= tb+satisfies _ _ _ = False++outside :: Double -> Cmp -> Double -> Bool+outside t CGt tb = t <= tb+outside t CGeq tb = t < tb+outside t CLt tb = t >= tb+outside t CLeq tb = t > tb+outside _ _ _ = False++-- ---------------------------------------------------------------------------+-- Path-condition entailment for fitted-tree pruning.+-- ---------------------------------------------------------------------------++-- | A known same-column threshold fact accumulated along a tree path.+data PredFact = PredFact !String !Cmp !Double++-- | The fact a branch's true edge establishes (the condition holds).+factTrue :: Expr Bool -> Maybe PredFact+factTrue e = (\a -> PredFact (aKey a) (aCmp a) (aThr a)) <$> atomOf e++{- | The fact a branch's false edge establishes (the negated condition). Only+sound for non-NaN (integral) columns — a NaN row takes the false edge too,+so @¬(x>t)@ is not a clean @x<=t@ bound for floats.+-}+factFalse :: Expr Bool -> Maybe PredFact+factFalse e = do+ a <- atomOf e+ guard (aIntegral a && aNull a == Total)+ nc <- negCmp (aCmp a)+ pure (PredFact (aKey a) nc (aThr a))++negCmp :: Cmp -> Maybe Cmp+negCmp CLt = Just CGeq+negCmp CLeq = Just CGt+negCmp CGt = Just CLeq+negCmp CGeq = Just CLt+negCmp _ = Nothing++{- | @entails facts cond@: 'Just' 'True' when the path facts force @cond@ true,+'Just' 'False' when they force it false, 'Nothing' when undecided.+-}+entails :: [PredFact] -> Expr Bool -> Maybe Bool+entails facts cond = do+ a <- atomOf cond+ let decisions =+ [ d+ | PredFact fk fc ft <- facts+ , fk == aKey a+ , Just d <- [factImplies (fc, ft) (aCmp a, aThr a)]+ ]+ case decisions of+ (d : _) -> Just d+ [] -> Nothing++{- | Does the fact's solution set sit inside @cond@ ('Just' 'True'), disjoint+from it ('Just' 'False'), or neither ('Nothing')? Boundary strictness is+honoured: e.g. @x<=t@ does NOT entail @x<t@, and @x>=t ∧ x<=t@ is not empty.+-}+factImplies :: (Cmp, Double) -> (Cmp, Double) -> Maybe Bool+factImplies (fc, ft) (cc, tc)+ | isLower fc, isLower cc, subset = Just True+ | isUpper fc, isUpper cc, subset = Just True+ | isLower fc, isUpper cc, disjointAtEq = Just False+ | isUpper fc, isLower cc, disjointBelow = Just False+ | otherwise = Nothing+ where+ fIncl = fc == CGeq || fc == CLeq+ cIncl = cc == CGeq || cc == CLeq+ subset =+ (if isLower fc then ft > tc else ft < tc)+ || (ft == tc && (not fIncl || cIncl))+ disjointAtEq = ft > tc || (ft == tc && not (fIncl && cIncl))+ disjointBelow = ft < tc || (ft == tc && not (fIncl && cIncl))
src-internal/DataFrame/Internal/Grouping.hs view
@@ -25,37 +25,53 @@ import Control.Exception (throw) import Control.Monad import Control.Monad.ST (ST, runST)+import Data.Bits (unsafeShiftR, (.&.)) import Data.Type.Equality (TestEquality (..), type (:~:) (Refl)) import DataFrame.Errors+import DataFrame.Internal.Algorithms.Hash+import DataFrame.Internal.Algorithms.Rank.Radix (rankByHash) import DataFrame.Internal.Column (- Bitmap, Column (..),- bitmapTestBit, materializeMerged, )-import DataFrame.Internal.DataFrame (DataFrame (..), GroupedDataFrame (..))-import DataFrame.Internal.DictEncode (dictEncodeColumnUpTo)-import DataFrame.Internal.GroupingDirect (- DirectGrouping (..),- directGroupThreshold,- tryDirectGroupColumn,+import DataFrame.Internal.Column.Bitmap (+ Bitmap,+ bitmapTestBit, )-import DataFrame.Internal.GroupingPar (parallelAssignGroups, shouldParallelize)-import DataFrame.Internal.Hash-import DataFrame.Internal.HashTable (htInsert, newHashTable)-import DataFrame.Internal.PackedText (+import DataFrame.Internal.Column.Encode (dictEncodeColumnUpTo)+import DataFrame.Internal.Column.Types+import DataFrame.Internal.Control.Concurrent (+ capabilities,+ chunksFor,+ forkJoin,+ parThreshold,+ shouldParallelize,+ splitChunkRange,+ )+import DataFrame.Internal.Data.HashTable (htInsert, newHashTable)+import DataFrame.Internal.Data.PackedText ( PackedSel, PackedTextData (..),- offAt, offCount, packedLength, packedSlice, selAt,- selLength, sliceEqBytes, )-import DataFrame.Internal.RadixRank (rankByHash)-import DataFrame.Internal.Types+import DataFrame.Internal.DataFrame (DataFrame (..), GroupedDataFrame (..))+import DataFrame.Internal.Grouping.Direct (+ DirectGrouping (..),+ ascendingCodeGroups,+ directGroupThreshold,+ rangeOf,+ tryDirectGroupColumn,+ )+import qualified DataFrame.Internal.Grouping.Direct as GD+import DataFrame.Internal.Grouping.Partitioned (+ parallelAssignGroups,+ rtgFromVisOffs,+ )+import DataFrame.Internal.Row.RowHash (computeRowHashesWithIO) import System.IO.Unsafe (unsafePerformIO) import Type.Reflection (typeRep) @@ -82,67 +98,167 @@ (VU.fromList [0]) VU.empty | Just dg <- tryDirectGroup names df = dg- | shouldParallelize n = groupByPar names df+ | shouldParallelize parThreshold n = groupByPar names df | otherwise = groupBySeq names df where !n = nRows df {- | Low-cardinality direct-indexed grouping fast path-('DataFrame.Internal.GroupingDirect'): fires only for a single clean small-range-@Int@ key. Returns 'Nothing' on any other key shape, falling back to the hash path.+('DataFrame.Internal.Grouping.Direct'): fires for key lists+where every key is a single clean small-range @Int@ column or a canonical+dict-encoded text column, and the product of the key domains stays within+'directGroupThreshold' (the keys fuse into one mixed-radix code; a single key+degenerates to its own code). Returns 'Nothing' on any other key shape, falling+back to the hash path.++Narrow domains build @offsets@/@groupRepRows@ eagerly (histogram-sized work+only) and leave BOTH per-row outputs lazy: @valueIndices@ ('visFromCodes')+only materializes for consumers that gather (median/top-k, set ops,+interpreter slices), and @rowToGroup@ ('rtgFromCodes') only for the streaming+scatter aggregations — each aggregate pays for exactly one O(n) output pass,+not both. Wide domains run the two-level radix engine instead+('DataFrame.Internal.Grouping.Direct.directLayoutLazy'): @rowToGroup@ eager,+@valueIndices@ deferred (see 'fusedDirectGroup'). -} tryDirectGroup :: [T.Text] -> DataFrame -> Maybe GroupedDataFrame-tryDirectGroup [name] df = do- col <- M.lookup name (columnIndices df) >>= \i -> columns df V.!? i- case tryDirectGroupColumn col of- Just dg ->- Just (Grouped df [name] (dgValueIndices dg) (dgOffsets dg) (dgRowToGroup dg))- Nothing -> case col of- PackedText Nothing p -> dictCodesGroup df [name] p- _ -> tryDictGroup (nRows df) df [name] col-tryDirectGroup _ _ = Nothing+tryDirectGroup [] _ = Nothing+tryDirectGroup names df = do+ cols <-+ traverse (\nm -> M.lookup nm (columnIndices df) >>= (columns df V.!?)) names+ case traverse fusedKey cols of+ Just keys -> fusedDirectGroup names df keys+ Nothing -> case (names, cols) of+ ([name], [col]) -> tryDictGroup (nRows df) df [name] col+ _ -> Nothing -dictCodesGroup ::- DataFrame -> [T.Text] -> PackedTextData -> Maybe GroupedDataFrame-dictCodesGroup df names p = do- sel <- ptSel p- guard (ptCanonicalSel p)- let offs = ptOffsets p- card = offCount offs - 1- n = selLength sel- guard (card > 0 && card <= directGroupThreshold && n > 0)- counts <- codeHistogram card sel- let occupied = VU.filter (\g -> VU.unsafeIndex counts g > 0) (VU.enumFromN 0 card)- nGroups = VU.length occupied- entryHash g =- let o = offAt offs g- in mixBytes fnvOffset (ptBytes p) o (offAt offs (g + 1) - o)- rank =- runST (rankByHash (pure . entryHash . VU.unsafeIndex occupied) nGroups)- remap = runST $ do- m <- VUM.new card- VU.imapM_ (\j g -> VUM.unsafeWrite m g (VU.unsafeIndex rank j)) occupied- VU.unsafeFreeze m- rtg = VU.generate n (VU.unsafeIndex remap . selAt sel)- (vis, os) = indicesFromGroups rtg nGroups- pure (Grouped df names vis os rtg)+{- | One key column of a fused multi-key direct grouping: a per-row component+code in @[0, fkDomain)@ (negative marks an invalid/corrupt code, which aborts+the direct path).+-}+data FusedKey = FusedKey+ { fkCode :: Int -> Int+ , fkDomain :: !Int+ } --- | Per-code occupancy counts; 'Nothing' as soon as any code is negative.-codeHistogram :: Int -> PackedSel -> Maybe (VU.Vector Int)-codeHistogram card sel = runST $ do- counts <- VUM.replicate card 0- let n = selLength sel- go i- | i >= n = Just <$> VU.unsafeFreeze counts- | otherwise = do- let c = selAt sel i- if c < 0 || c >= card- then pure Nothing- else do- VUM.unsafeModify counts (+ 1) c- go (i + 1)- go 0+{- | Classify a key column for the fused multi-key direct path: a clean non-null+unboxed @Int@ of small range, or a non-null canonical dict-encoded text column+of small dictionary. Anything else falls back to the hash group-by.+-}+fusedKey :: Column -> Maybe FusedKey+fusedKey (UnboxedColumn Nothing (v :: VU.Vector a))+ | Just Refl <- testEquality (typeRep @a) (typeRep @Int)+ , not (VU.null v) =+ let (!mn, !mx) = rangeOf v+ !range = mx - mn + 1+ in if range >= 1 && range <= directGroupThreshold+ then Just (FusedKey (\i -> VU.unsafeIndex v i - mn) range)+ else Nothing+fusedKey (PackedText Nothing p)+ | Just sel <- ptSel p+ , ptCanonicalSel p =+ let offs = ptOffsets p+ !card = offCount offs - 1+ in if card >= 1 && card <= directGroupThreshold+ then+ Just+ ( FusedKey+ (\i -> let c = selAt sel i in if c >= card then -1 else c)+ card+ )+ else Nothing+fusedKey _ = Nothing +{- | Fuse the per-key codes into one mixed-radix code per row+(@((k1*d2)+k2)*d3+...@) and feed the direct counting-sort machinery. Group+order: ascending fused code (lexicographic in key order) — ascending value+order for @Int@ keys (mirroring the order the single-@Int@-key direct path+always had) and ascending dictionary code for dict-encoded text keys (the+dictionary's first-appearance order, a fixed property of the column). The+ascending order keeps @codeToGroup@ an identity map whenever the domain is+fully occupied, so the deferred @rowToGroup@ pass skips its per-row random+remap lookup; ranking dict groups by string hash instead (the historical+order) profiled ~0.6s slower per 1e8 rows at 1e6 groups.++On the narrow-domain engine, @valueIndices@ and @rowToGroup@ are passed to the+constructor as unevaluated applications of 'visFromCodes' / 'rtgFromCodes'+(constructor arguments are not forced even under @-XStrict@, and the fields+are lazy at their definition site), so each per-row output pass is deferred+until a consumer demands it. The wide-domain engine defers only+@valueIndices@ (see the branch comment below).+-}+fusedDirectGroup ::+ [T.Text] -> DataFrame -> [FusedKey] -> Maybe GroupedDataFrame+fusedDirectGroup names df keys = do+ domain <- fusedDomain (map fkDomain keys)+ let n = nRows df+ codeAt' = fusedCodeAt keys+ if GD.useTwoLevel n domain+ then do+ {- Wide domains (> ~1024 codes at parallel scale): the two-level+ radix engine — no pass random-writes a multi-megabyte table per+ worker, unlike the per-chunk direct histograms below (measured ~2x+ on the eager layout at 1e6 codes / 1e8 rows). It builds+ @rowToGroup@ eagerly (the streaming aggregations force it first+ thing anyway); only @valueIndices@ stays deferred, reconstructed+ from @rowToGroup@ by the same engine on demand. -}+ (rtg, offs, reps, nGroups) <-+ GD.directLayoutLazy codeAt' n domain ascendingCodeGroups+ Just+ ( GroupedInternal+ df+ names+ (GD.visFromRowToGroup n nGroups offs rtg)+ offs+ rtg+ reps+ )+ else do+ (offs, reps, counts, ctg, hists, nGroups) <-+ directLayoutLazy codeAt' n domain ascendingCodeGroups+ Just+ ( GroupedInternal+ df+ names+ (visFromCodes codeAt' counts ctg offs hists n domain)+ offs+ (rtgFromCodes codeAt' ctg n)+ reps+ )++{- | Product of the per-key domains, 'Nothing' once it (or any factor) passes+'directGroupThreshold'. Factors are capped before multiplying, so the running+product never exceeds @threshold^2@ and cannot overflow.+-}+fusedDomain :: [Int] -> Maybe Int+fusedDomain = go 1+ where+ go !acc [] = Just acc+ go !acc (d : ds)+ | d < 1 || d > directGroupThreshold = Nothing+ | acc * d > directGroupThreshold = Nothing+ | otherwise = go (acc * d) ds++{- | Per-row fused mixed-radix code; @-1@ when any component code is invalid+(only possible for corrupt dict codes), making 'groupCodesMaybe' bail to the+hash path. Valid components compose to a code in @[0, product of domains)@.+-}+fusedCodeAt :: [FusedKey] -> (Int -> Int)+fusedCodeAt [] = const (-1)+fusedCodeAt (k0 : ks0) = go (fkCode k0) ks0+ where+ go f [] = f+ go f (k : ks) =+ let !d = fkDomain k+ g = fkCode k+ in go+ ( \i ->+ let a = f i+ in if a < 0+ then -1+ else let b = g i in if b < 0 then -1 else a * d + b+ )+ ks+ {- | Dictionary-encode a single text key to dense int codes, then derive @valueIndices@/@offsets@ by counting sort. Profiled slower than the fused hash group-by on every db-benchmark question, so it always falls back ('dictGroupEnabled').@@ -150,7 +266,7 @@ tryDictGroup :: Int -> DataFrame -> [T.Text] -> Column -> Maybe GroupedDataFrame tryDictGroup n df names col- | dictGroupEnabled && not (shouldParallelize n) = do+ | dictGroupEnabled && not (shouldParallelize parThreshold n) = do (codes, card) <- dictEncodeColumnUpTo dictSingleThreshold col let (vis, os) = indicesFromGroups codes card Just (Grouped df names vis os codes)@@ -184,7 +300,7 @@ (vis, os) = indicesFromGroups rtg nGroups in Grouped df names vis os rtg -{- | The parallel partitioned grouping path (see 'DataFrame.Internal.GroupingPar'):+{- | The parallel partitioned grouping path (see 'DataFrame.Internal.Grouping.Partitioned'): forks one task per capability, producing output bit-for-bit identical to 'groupBySeq'. Pure via 'unsafePerformIO' (deterministic thread fan-out only). -}@@ -192,10 +308,20 @@ groupByPar names df = let !n = nRows df indicesToGroup = keyColIndices names df- !hashes = runST (computeHashes df indicesToGroup n)+ -- Merged key columns are exotic; hash their eager form.+ selectedCols = map (materializeMerged . (columns df V.!)) indicesToGroup !eqRow = eqKeyRow df indicesToGroup- (rtg, vis, os) = unsafePerformIO (parallelAssignGroups n hashes eqRow)- in Grouped df names vis os rtg+ (vis, os) = unsafePerformIO $ do+ -- Parallel row-hash kernel, bit-identical to 'computeHashes' at the+ -- same dict-code setting (grouping always hashes canonical dict+ -- columns by code; see 'hashPacked').+ hashes <- computeRowHashesWithIO True n selectedCols+ parallelAssignGroups n hashes eqRow+ in -- rowToGroup is passed as an UNFORCED constructor argument (this module+ -- is -XStrict, so it must not be let-bound): gather-style aggregation+ -- over huge group counts never reads it, and the deferred pass writes+ -- values identical to the eager build.+ Grouped df names vis os (rtgFromVisOffs n vis os) {-# NOINLINE groupByPar #-} -- | Column indices of the requested key columns, in column order.@@ -316,10 +442,16 @@ let eqV a b = V.unsafeIndex v a == V.unsafeIndex v b in withNulls bm eqV colEqRow (PackedText bm p) =- let eqV a b =- let (arrA, oA, lA) = packedSlice p a- (arrB, oB, lB) = packedSlice p b- in sliceEqBytes arrA oA lA arrB oB lB+ -- A canonical dictionary selection assigns equal strings the same code,+ -- so two rows are byte-equal iff their codes agree.+ let eqV = case ptSel p of+ Just sel+ | ptCanonicalSel p ->+ \a b -> selAt sel a == selAt sel b+ _ -> \a b ->+ let (arrA, oA, lA) = packedSlice p a+ (arrB, oB, lB) = packedSlice p b+ in sliceEqBytes arrA oA lA arrB oB lB in withNulls bm eqV {-# INLINE colEqRow #-} @@ -380,6 +512,448 @@ frozenVis <- VU.unsafeFreeze vis pure (frozenVis, offs) +-------------------------------------------------------------------------------+-- Deferred-placement direct grouping+-------------------------------------------------------------------------------++{- | Contiguous per-worker row ranges: one chunk per capability above the+parallel threshold, a single chunk otherwise.+-}+directRowChunks :: Int -> [(Int, Int)]+directRowChunks = chunksFor parThreshold++{- | Like 'directRowChunks' but over a code/group domain (merge/seed passes),+which pays for a fan-out at a much lower width than the row passes do.+-}+directCodeSlices :: Int -> [(Int, Int)]+directCodeSlices = chunksFor 4096++{- | The eager phases of the direct counting-sort grouping — WITHOUT either+per-row output pass: per-chunk validated code histograms (parallel), per-code+totals and first occurrences (parallel over code slices), the caller-chosen+code->group mapping, the offsets prefix scan and per-group representative+rows. Returns+@(offsets, groupRepRows, counts, codeToGroup, chunkHists, nGroups)@ — the last+three feed the deferred @valueIndices@ placement ('visFromCodes') and+@rowToGroup@ ('rtgFromCodes') thunks, so a consumer pays only for the per-row+output it actually demands. 'Nothing' when any row's code falls outside+@[0, card)@ (fall back to hashing).++Pure w.r.t. its immutable inputs: the fork fan-out is a fixed function of the+row count and capability count, and every merge runs in fixed chunk order, so+the result is deterministic and the 'unsafePerformIO' is safe.+-}+directLayoutLazy ::+ (Int -> Int) ->+ Int ->+ Int ->+ (VU.Vector Int -> (VU.Vector Int, Int)) ->+ Maybe+ ( VU.Vector Int+ , VU.Vector Int+ , VU.Vector Int+ , VU.Vector Int+ , [VU.Vector Int]+ , Int+ )+directLayoutLazy codeAt n card mkGroups+ | n <= 0 || card <= 0 = Nothing+ | n < packedRowLimit = unsafePerformIO $ do+ -- Packed variant: count and first-occurrence row share one word per+ -- code, keeping phase 1 at a single accumulator array per chunk+ -- (measured ~0.15s/1e8 rows cheaper than a second firstOcc array).+ let chunks = directRowChunks n+ parts <- forkJoin [histFirstChunkPacked codeAt card lo hi | (lo, hi) <- chunks]+ if not (all snd parts)+ then pure Nothing+ else finishLayout card mkGroups (map fst parts) $ \histsM totalsM firstAllM lo hi ->+ sumFirstSlicePacked histsM totalsM firstAllM lo hi+ | otherwise = unsafePerformIO $ do+ -- Fallback for gigantic frames where a row index does not fit the+ -- packed word: separate count and firstOcc arrays, same results.+ let chunks = directRowChunks n+ parts <- forkJoin [histFirstChunk codeAt card lo hi | (lo, hi) <- chunks]+ if not (all (\(_, _, ok) -> ok) parts)+ then pure Nothing+ else+ finishLayout+ card+ mkGroups+ (map (\(h, _, _) -> h) parts)+ ( \histsM totalsM firstAllM lo hi ->+ sumFirstSlice histsM (map (\(_, f, _) -> f) parts) totalsM firstAllM lo hi+ )+{-# NOINLINE directLayoutLazy #-}++{- | Shared tail of 'directLayoutLazy': run the totals/first-occurrence merge+(which also normalizes each chunk histogram to plain counts, see+'sumFirstSlicePacked'), derive the group mapping, offsets and representative+rows, and freeze the retained chunk histograms.+-}+finishLayout ::+ Int ->+ (VU.Vector Int -> (VU.Vector Int, Int)) ->+ [VUM.IOVector Int] ->+ ( [VUM.IOVector Int] ->+ VUM.IOVector Int ->+ VUM.IOVector Int ->+ Int ->+ Int ->+ IO ()+ ) ->+ IO+ ( Maybe+ ( VU.Vector Int+ , VU.Vector Int+ , VU.Vector Int+ , VU.Vector Int+ , [VU.Vector Int]+ , Int+ )+ )+finishLayout card mkGroups histsM mergeSlice = do+ totalsM <- VUM.new card+ firstAllM <- VUM.new card+ _ <-+ forkJoin+ [ mergeSlice histsM totalsM firstAllM lo hi+ | (lo, hi) <- directCodeSlices card+ ]+ counts <- VU.unsafeFreeze totalsM+ firstAll <- VU.unsafeFreeze firstAllM+ let (codeToGroup, nGroups) = mkGroups counts+ offs <- scanGroupOffsets counts codeToGroup nGroups+ repsM <- VUM.new nGroups+ _ <-+ forkJoin+ [ scatterRepsSlice counts codeToGroup firstAll repsM lo hi+ | (lo, hi) <- directCodeSlices card+ ]+ reps <- VU.unsafeFreeze repsM+ hists <- mapM VU.unsafeFreeze histsM+ pure (Just (offs, reps, counts, codeToGroup, hists, nGroups))++{- | Rows must satisfy @row + 1 < 2^31@ for the packed count/first-row encoding+(count in the high bits, first row + 1 in the low 31). Above it (a >2e9-row+frame, >17GB per Int column) the unpacked variant runs instead.+-}+packedRowLimit :: Int+packedRowLimit = 0x7FFFFFFF++-- | One unit of count in the packed encoding; also the low-bits mask + 1.+packedCountOne :: Int+packedCountOne = 0x80000000++{- | Whether @codeToGroup@ maps every code to itself (fully occupied ascending+domain — e.g. a dense Int key covering its whole range). The rowToGroup pass+then skips the random remap lookup entirely.+-}+isIdentityMap :: VU.Vector Int -> Bool+isIdentityMap m = go 0+ where+ !k = VU.length m+ go !i+ | i >= k = True+ | VU.unsafeIndex m i /= i = False+ | otherwise = go (i + 1)++{- | Histogram one row chunk with the packed encoding: slot @c@ holds+@count(c) * 2^31 + (firstRow(c) + 1)@ (zero = never seen). One accumulator+array per chunk. Reports 'False' as soon as any code escapes @[0, card)@.+-}+histFirstChunkPacked ::+ (Int -> Int) -> Int -> Int -> Int -> IO (VUM.IOVector Int, Bool)+histFirstChunkPacked codeAt card lo hi = do+ acc <- VUM.replicate card (0 :: Int)+ let go !i+ | i >= hi = pure True+ | otherwise = do+ let !c = codeAt i+ if c < 0 || c >= card+ then pure False+ else do+ x <- VUM.unsafeRead acc c+ VUM.unsafeWrite+ acc+ c+ (if x == 0 then packedCountOne + (i + 1) else x + packedCountOne)+ go (i + 1)+ ok <- go lo+ pure (acc, ok)++{- | Per-code totals and overall first occurrences from the PACKED chunk+histograms, rewriting each histogram slot to its plain count in place (the+placement thunk then sees ordinary counts). Chunks are ordered by row range, so+the first chunk with a nonzero slot holds the code's globally first row.+-}+sumFirstSlicePacked ::+ [VUM.IOVector Int] ->+ VUM.IOVector Int ->+ VUM.IOVector Int ->+ Int ->+ Int ->+ IO ()+sumFirstSlicePacked hists totals firstAll lo hi = go lo+ where+ go !c+ | c >= hi = pure ()+ | otherwise = do+ let sumP [] !acc !firstRow = pure (acc, firstRow)+ sumP (h : hs) !acc !firstRow = do+ x <- VUM.unsafeRead h c+ let !cnt = x `unsafeShiftR` 31+ VUM.unsafeWrite h c cnt+ if firstRow < 0 && x /= 0+ then sumP hs (acc + cnt) ((x .&. (packedCountOne - 1)) - 1)+ else sumP hs (acc + cnt) firstRow+ (s, fo) <- sumP hists 0 (-1)+ VUM.unsafeWrite totals c s+ VUM.unsafeWrite firstAll c fo+ go (c + 1)++{- | Histogram one row chunk into a private @card@-slot count plus the chunk's+first occurrence of each code, reporting 'False' as soon as any code escapes+@[0, card)@ (the counts are then abandoned). Fallback for frames beyond+'packedRowLimit'.+-}+histFirstChunk ::+ (Int -> Int) ->+ Int ->+ Int ->+ Int ->+ IO (VUM.IOVector Int, VUM.IOVector Int, Bool)+histFirstChunk codeAt card lo hi = do+ acc <- VUM.replicate card (0 :: Int)+ firstOcc <- VUM.replicate card (-1 :: Int)+ let go !i+ | i >= hi = pure True+ | otherwise = do+ let !c = codeAt i+ if c < 0 || c >= card+ then pure False+ else do+ x <- VUM.unsafeRead acc c+ VUM.unsafeWrite acc c (x + 1)+ when (x == 0) (VUM.unsafeWrite firstOcc c i)+ go (i + 1)+ ok <- go lo+ pure (acc, firstOcc, ok)++{- | Per-code totals over one code slice, plus the overall first occurrence of+each code: the chunks are ordered by row range, so the first chunk with a+nonzero count for a code holds its globally first row.+-}+sumFirstSlice ::+ [VUM.IOVector Int] ->+ [VUM.IOVector Int] ->+ VUM.IOVector Int ->+ VUM.IOVector Int ->+ Int ->+ Int ->+ IO ()+sumFirstSlice hists firsts totals firstAll lo hi = go lo+ where+ go !c+ | c >= hi = pure ()+ | otherwise = do+ let sumP [] [] !acc !firstRow = pure (acc, firstRow)+ sumP (h : hs) (f : fs) !acc !firstRow = do+ x <- VUM.unsafeRead h c+ if firstRow < 0 && x > 0+ then do+ fo <- VUM.unsafeRead f c+ sumP hs fs (acc + x) fo+ else sumP hs fs (acc + x) firstRow+ sumP _ _ _ _ = error "sumFirstSlice: mismatched partials"+ (s, fo) <- sumP hists firsts 0 (-1)+ VUM.unsafeWrite totals c s+ VUM.unsafeWrite firstAll c fo+ go (c + 1)++{- | Exclusive prefix scan of per-group counts (gathered through @codeToGroup@)+into the offsets array of length @nGroups + 1@.+-}+scanGroupOffsets :: VU.Vector Int -> VU.Vector Int -> Int -> IO (VU.Vector Int)+scanGroupOffsets counts codeToGroup nGroups = do+ let !card = VU.length counts+ grpCount <- VUM.new nGroups+ let gather !c+ | c >= card = pure ()+ | otherwise = do+ let !cnt = VU.unsafeIndex counts c+ if cnt == 0+ then gather (c + 1)+ else do+ VUM.unsafeWrite grpCount (VU.unsafeIndex codeToGroup c) cnt+ gather (c + 1)+ gather 0+ offsM <- VUM.new (nGroups + 1)+ let scan !g !acc+ | g >= nGroups = VUM.unsafeWrite offsM nGroups acc+ | otherwise = do+ VUM.unsafeWrite offsM g acc+ c <- VUM.unsafeRead grpCount g+ scan (g + 1) (acc + c)+ scan 0 0+ VU.unsafeFreeze offsM++{- | @reps[codeToGroup c] = firstAll c@ for every occupied code: each group is+exactly one occupied code, so this is a disjoint parallel write and equals+@vis[offs[g]]@ (the group's first row in original order).+-}+scatterRepsSlice ::+ VU.Vector Int ->+ VU.Vector Int ->+ VU.Vector Int ->+ VUM.IOVector Int ->+ Int ->+ Int ->+ IO ()+scatterRepsSlice counts codeToGroup firstAll repsM lo hi = go lo+ where+ go !c+ | c >= hi = pure ()+ | VU.unsafeIndex counts c == 0 = go (c + 1)+ | otherwise = do+ VUM.unsafeWrite+ repsM+ (VU.unsafeIndex codeToGroup c)+ (VU.unsafeIndex firstAll c)+ go (c + 1)++{- | Deferred @rowToGroup@: one parallel per-row pass mapping each row's code+through @codeToGroup@ (skipping the lookup entirely when the map is the+identity, i.e. a fully occupied ascending domain). Only the streaming+aggregation paths force this; a purely gather-driven consumer (median, top-k)+never pays for it.++Pure w.r.t. its immutable inputs and deterministic (fixed chunking), so the+'unsafePerformIO' behind a lazy field is safe: whenever and however many times+the thunk is forced it yields the same vector.+-}+rtgFromCodes :: (Int -> Int) -> VU.Vector Int -> Int -> VU.Vector Int+rtgFromCodes codeAt codeToGroup n = unsafePerformIO $ do+ rtgM <- VUM.new n+ let identity = isIdentityMap codeToGroup+ _ <-+ forkJoin+ [ ( if identity+ then rtgChunkIdentity codeAt rtgM lo hi+ else rtgChunk codeAt codeToGroup rtgM lo hi+ )+ | (lo, hi) <- directRowChunks n+ ]+ VU.unsafeFreeze rtgM+{-# NOINLINE rtgFromCodes #-}++-- | @rtg[i] = codeToGroup (codeAt i)@ over one row chunk (disjoint writes).+rtgChunk ::+ (Int -> Int) -> VU.Vector Int -> VUM.IOVector Int -> Int -> Int -> IO ()+rtgChunk codeAt codeToGroup rtgM lo hi = go lo+ where+ go !i+ | i >= hi = pure ()+ | otherwise = do+ VUM.unsafeWrite rtgM i (VU.unsafeIndex codeToGroup (codeAt i))+ go (i + 1)++-- | 'rtgChunk' without the remap lookup (codeToGroup is the identity).+rtgChunkIdentity ::+ (Int -> Int) -> VUM.IOVector Int -> Int -> Int -> IO ()+rtgChunkIdentity codeAt rtgM lo hi = go lo+ where+ go !i+ | i >= hi = pure ()+ | otherwise = do+ VUM.unsafeWrite rtgM i (codeAt i)+ go (i + 1)++{- | Deferred stable placement: build the @valueIndices@ permutation from the+per-row codes and the RETAINED phase-1 chunk histograms — the same+seed-cursors-then-place structure (and cost) the eager path used, minus the+@rowToGroup@ writes. Each chunk's code-indexed cursor starts at the group+offset plus everything earlier chunks (in row order) place there, so rows keep+original order within each group: the result is the unique group-major,+original-row-order permutation, bit-identical to the eager placement at any+chunk count.++Pure w.r.t. its immutable inputs and deterministic (fixed chunking, fixed merge+order), so the 'unsafePerformIO' behind a lazy field is safe: whenever and+however many times the thunk is forced it yields the same vector. The thunk+retains the chunk histograms (capabilities x card words) until forced or the+grouping is dropped.+-}+visFromCodes ::+ (Int -> Int) ->+ VU.Vector Int ->+ VU.Vector Int ->+ VU.Vector Int ->+ [VU.Vector Int] ->+ Int ->+ Int ->+ VU.Vector Int+visFromCodes codeAt counts codeToGroup offs hists n card = unsafePerformIO $ do+ let chunks = directRowChunks n+ -- Private mutable copies of the retained histograms, rewritten in place+ -- into the per-chunk write cursors.+ cursors <- mapM VU.thaw hists+ _ <-+ forkJoin+ [ seedCursorSlice counts codeToGroup offs cursors lo hi+ | (lo, hi) <- directCodeSlices card+ ]+ vis <- VUM.new n+ _ <-+ forkJoin+ [ placeVisChunk codeAt cursor vis lo hi+ | ((lo, hi), cursor) <- zip chunks cursors+ ]+ VU.unsafeFreeze vis+{-# NOINLINE visFromCodes #-}++{- | Rewrite each chunk's histogram copy in place into its disjoint write+cursor: chunk w's run for code c starts at the offset of c's group plus what+earlier chunks (in row order) place there.+-}+seedCursorSlice ::+ VU.Vector Int ->+ VU.Vector Int ->+ VU.Vector Int ->+ [VUM.IOVector Int] ->+ Int ->+ Int ->+ IO ()+seedCursorSlice counts codeToGroup offs cursors lo hi = go lo+ where+ go !c+ | c >= hi = pure ()+ | VU.unsafeIndex counts c == 0 = go (c + 1)+ | otherwise = do+ let !g = VU.unsafeIndex codeToGroup c+ loop [] !_ = pure ()+ loop (cur : rest) !acc = do+ t <- VUM.unsafeRead cur c+ VUM.unsafeWrite cur c acc+ loop rest (acc + t)+ loop cursors (VU.unsafeIndex offs g)+ go (c + 1)++{- | Stable placement over one row chunk via the chunk's advancing+code-indexed cursors.+-}+placeVisChunk ::+ (Int -> Int) -> VUM.IOVector Int -> VUM.IOVector Int -> Int -> Int -> IO ()+placeVisChunk codeAt cursor vis lo hi = go lo+ where+ go !i+ | i >= hi = pure ()+ | otherwise = do+ let !c = codeAt i+ pos <- VUM.unsafeRead cursor c+ VUM.unsafeWrite vis pos i+ VUM.unsafeWrite cursor c (pos + 1)+ go (i + 1)+ {- | Fold a value-mix over an unboxed column into the running hash vector, respecting the null bitmap: a null slot mixes a fixed 'nullSalt' sentinel. -}@@ -412,11 +986,17 @@ {- | Hash a packed-text column over its raw UTF-8 byte slices (no per-row 'Data.Text.Text'), mixing 'nullSalt' for null rows. Shares 'mixBytes' with-'mixText' so packed and boxed Text columns hash identically.+'mixText' so packed and boxed Text columns hash identically. A canonical+dict-encoded column (equal strings share a code) instead mixes its 'Int' code+with one 'mixInt' per row; 'DataFrame.Internal.RowHash.packedRange' applies the+same rule under the grouping setting so 'groupBySeq' and 'groupByPar' bucket+identically. -} hashPacked :: VUM.MVector s Int -> Maybe Bitmap -> PackedTextData -> ST s ()-hashPacked mh bm p = go 0+hashPacked mh bm p = case ptSel p of+ Just sel | ptCanonicalSel p -> goCodes sel 0+ _ -> go 0 where !n = packedLength p go !i@@ -428,6 +1008,15 @@ _ -> let (arr, o, l) = packedSlice p i in mixBytes h arr o l VUM.unsafeWrite mh i h' go (i + 1)+ goCodes !sel !i+ | i >= n = pure ()+ | otherwise = do+ !h <- VUM.unsafeRead mh i+ let h' = case bm of+ Just bm' | not (bitmapTestBit bm' i) -> mixInt h nullSalt+ _ -> mixInt h (selAt sel i)+ VUM.unsafeWrite mh i h'+ goCodes sel (i + 1) {-# INLINE hashPacked #-} -- Inline accessors to avoid depending on Operations.Core
+ src-internal/DataFrame/Internal/Grouping/Direct.hs view
@@ -0,0 +1,859 @@+{-# LANGUAGE BangPatterns #-}+{-# LANGUAGE ExplicitNamespaces #-}+{-# LANGUAGE GADTs #-}+{-# LANGUAGE ScopedTypeVariables #-}+{-# LANGUAGE TypeApplications #-}++{- | Low-cardinality direct-indexed grouping fast path: when every row's key+reduces to a dense @Int@ code in a small domain, the code itself indexes a dense+accumulator (no hashing/probing). All O(n) passes run chunked across+capabilities — per-chunk histograms feed prefix-summed disjoint write cursors —+so the stable within-group row order of the sequential counting sort is+reproduced exactly.++Two engines cover the code-domain spectrum with bit-identical results:++* narrow domains (@card <= 'twoLevelCardThreshold'@) index per-worker+ histogram\/cursor tables directly — they stay cache-resident;++* wide domains use a two-level radix split (top code bits pick one of ~@2^10@+ buckets, cursors cache-resident) so no pass ever random-writes a+ multi-megabyte table per worker.++'directLayoutLazy' is the aggregation entry point: it builds @rowToGroup@,+@offsets@ and the per-group representative rows eagerly but skips the O(n)+stable placement entirely; 'visFromRowToGroup' reconstructs @valueIndices@+on demand (its value is the unique stable counting-sort permutation, so WHEN it+runs is unobservable).+-}+module DataFrame.Internal.Grouping.Direct (+ directGroupThreshold,+ tryDirectGroupColumn,+ groupCodesMaybe,+ directLayoutLazy,+ visFromRowToGroup,+ ascendingCodeGroups,+ rangeOf,+ useTwoLevel,+ DirectGrouping (..),+) where++import Data.Type.Equality (TestEquality (..), type (:~:) (Refl))+import qualified Data.Vector.Unboxed as VU+import qualified Data.Vector.Unboxed.Mutable as VUM+import System.IO.Unsafe (unsafePerformIO)+import Type.Reflection (typeRep)++import Control.Monad (when)+import Control.Monad.ST (runST)+import Data.Bits (countLeadingZeros, unsafeShiftL, unsafeShiftR, (.&.))+import DataFrame.Internal.Column (Column (..))+import DataFrame.Internal.Control.Concurrent (+ capabilities,+ chunksFor,+ forkJoin,+ parThreshold,+ parallelChunks,+ pooledIndices,+ shouldParallelize,+ splitChunkRange,+ )++{- | Largest key code DOMAIN (single-key value range, or the product of per-key+domains for a fused multi-key code) the direct grouping path accepts. A @2^20@-slot+histogram is 8MB; the low-cardinality questions sit far below it (id4 range 100,+id6 range 1e5). Wider domains fall back to the hash group-by.+-}+directGroupThreshold :: Int+directGroupThreshold = 1048576++{- | The grouping layout the hash path also produces: @rowToGroup@, the+group-sorted @valueIndices@, the @offsets@ prefix array, and the group count.+-}+data DirectGrouping = DirectGrouping+ { dgRowToGroup :: !(VU.Vector Int)+ , dgValueIndices :: !(VU.Vector Int)+ , dgOffsets :: !(VU.Vector Int)+ , dgNGroups :: !Int+ }++{- | Take the direct path if the (single) key column is a clean non-null unboxed+@Int@ column with a small value range. Returns 'Nothing' to fall back to the+hash group-by on anything else (boxed/text keys, nullable, wide ranges, empty).+-}+tryDirectGroupColumn :: Column -> Maybe DirectGrouping+tryDirectGroupColumn (UnboxedColumn Nothing (v :: VU.Vector a))+ | Just Refl <- testEquality (typeRep @a) (typeRep @Int)+ , not (VU.null v) =+ let (!mn, !mx) = rangeOf v+ !range = mx - mn + 1+ in if range >= 1 && range <= directGroupThreshold+ then+ groupCodesMaybe+ (\i -> VU.unsafeIndex v i - mn)+ (VU.length v)+ range+ ascendingCodeGroups+ else Nothing+tryDirectGroupColumn _ = Nothing++-- | Parallel min/max reduce (order-independent).+rangeOf :: VU.Vector Int -> (Int, Int)+rangeOf v+ | not (shouldParallelize parThreshold n) = rangeChunk v 0 n+ | otherwise = unsafePerformIO $ do+ rs <- parallelChunks parThreshold n (\lo hi -> pure $! rangeChunk v lo hi)+ pure (combineRanges (filter (\(a, _) -> a /= maxBound) rs))+ where+ !n = VU.length v+{-# NOINLINE rangeOf #-}++rangeChunk :: VU.Vector Int -> Int -> Int -> (Int, Int)+rangeChunk v lo hi = go lo maxBound minBound+ where+ go !i !mn !mx+ | i >= hi = (mn, mx)+ | otherwise =+ let !x = VU.unsafeIndex v i+ in go (i + 1) (min mn x) (max mx x)++combineRanges :: [(Int, Int)] -> (Int, Int)+combineRanges [] = (0, 0)+combineRanges ((a0, b0) : rest) = foldr (\(a, b) (ma, mb) -> (min ma a, max mb b)) (a0, b0) rest++-- | Whether to fan out at this row count.+shouldPar :: Int -> Bool+shouldPar = shouldParallelize parThreshold++{- | Contiguous per-worker row ranges: one chunk per capability above the+parallel threshold, a single chunk otherwise (the sequential fallback runs the+same code on the calling thread).+-}+rowChunks :: Int -> [(Int, Int)]+rowChunks = chunksFor parThreshold++{- | Like 'rowChunks' but over the code domain (parallel merge/seed passes),+which pays for a fan-out at a much lower width than the row passes do.+-}+codeSlices :: Int -> [(Int, Int)]+codeSlices = chunksFor 4096++{- | Run each action on its own thread and collect the results in order;+rethrow the first failure. A single action runs on the calling thread.+-}++-------------------------------------------------------------------------------+-- Full grouping (eager valueIndices): compatibility entry point+-------------------------------------------------------------------------------++{- | Build the grouping by counting sort on a per-row code in @[0, card)@.+Returns 'Nothing' when any row's code falls outside @[0, card)@ (fall back to+hashing).++@mkGroups counts@ must return a dense group id for every code with a nonzero+count (other slots are never read; occupied codes must get distinct ids) and+the group count; it decides group order.++Equivalent to 'directLayoutLazy' plus a forced 'visFromRowToGroup'; kept for+callers that want the whole layout eagerly.+-}+groupCodesMaybe ::+ (Int -> Int) ->+ Int ->+ Int ->+ (VU.Vector Int -> (VU.Vector Int, Int)) ->+ Maybe DirectGrouping+groupCodesMaybe codeAt n card mkGroups = do+ (rtg, offs, _reps, nGroups) <- directLayoutLazy codeAt n card mkGroups+ let !vis = visFromRowToGroup n nGroups offs rtg+ Just (DirectGrouping rtg vis offs nGroups)++-------------------------------------------------------------------------------+-- Eager layout without placement: rowToGroup + offsets + group rep rows+-------------------------------------------------------------------------------++{- | The layout every aggregation needs, WITHOUT the O(n) stable placement:+@(rowToGroup, offsets, groupRepRows, nGroups)@, all four computed eagerly.+@groupRepRows[g]@ is the first original row of group @g@ (what+@valueIndices[offsets[g]]@ evaluates to). Pair with 'visFromRowToGroup' for a+deferred @valueIndices@. Returns 'Nothing' when any row's code falls outside+@[0, card)@.++@mkGroups@ contract as in 'groupCodesMaybe'.+-}+directLayoutLazy ::+ (Int -> Int) ->+ Int ->+ Int ->+ (VU.Vector Int -> (VU.Vector Int, Int)) ->+ Maybe (VU.Vector Int, VU.Vector Int, VU.Vector Int, Int)+directLayoutLazy codeAt n card mkGroups+ | n <= 0 || card <= 0 = Nothing+ | useTwoLevel n card = unsafePerformIO (layoutWide codeAt n card mkGroups)+ | otherwise = unsafePerformIO (layoutNarrow codeAt n card mkGroups)+{-# NOINLINE directLayoutLazy #-}++{- | Narrow domains (and the sequential small-@n@ fallback): per-chunk direct+histograms with first-occurrence tracking, merged over code slices.+-}+layoutNarrow ::+ (Int -> Int) ->+ Int ->+ Int ->+ (VU.Vector Int -> (VU.Vector Int, Int)) ->+ IO (Maybe (VU.Vector Int, VU.Vector Int, VU.Vector Int, Int))+layoutNarrow codeAt n card mkGroups = do+ let chunks = rowChunks n+ parts <- forkJoin [histFirstChunk codeAt card lo hi | (lo, hi) <- chunks]+ if not (all (\(_, _, ok) -> ok) parts)+ then pure Nothing+ else do+ let partials = [(cs, fs) | (cs, fs, _) <- parts]+ totalsM <- VUM.unsafeNew card+ firstRowM <- VUM.unsafeNew card+ _ <-+ forkJoin+ [mergeSlice partials totalsM firstRowM lo hi | (lo, hi) <- codeSlices card]+ counts <- VU.unsafeFreeze totalsM+ firstRow <- VU.unsafeFreeze firstRowM+ finishLayout codeAt n card mkGroups counts firstRow++{- | Histogram one row chunk into a private @card@-slot count, recording the+chunk's first row of each code, and reporting invalid codes (third component+'False'; the arrays are then abandoned).+-}+histFirstChunk ::+ (Int -> Int) ->+ Int ->+ Int ->+ Int ->+ IO (VUM.IOVector Int, VUM.IOVector Int, Bool)+histFirstChunk codeAt card lo hi = do+ acc <- VUM.replicate card (0 :: Int)+ firstV <- VUM.unsafeNew card+ let go !i+ | i >= hi = pure True+ | otherwise = do+ let !c = codeAt i+ if c < 0 || c >= card+ then pure False+ else do+ x <- VUM.unsafeRead acc c+ when (x == 0) (VUM.unsafeWrite firstV c i)+ VUM.unsafeWrite acc c (x + 1)+ go (i + 1)+ ok <- go lo+ pure (acc, firstV, ok)++{- | @totals[c] = Σ_w counts_w[c]@ and @firstRow[c]@ = the first chunk's first+occurrence (chunks are in row order, so that IS the global first row of @c@),+over one code slice. Every @totals@ slot is written; @firstRow[c]@ only where+the count is nonzero (never read otherwise).+-}+mergeSlice ::+ [(VUM.IOVector Int, VUM.IOVector Int)] ->+ VUM.IOVector Int ->+ VUM.IOVector Int ->+ Int ->+ Int ->+ IO ()+mergeSlice partials totals firstRow lo hi = go lo+ where+ go !c+ | c >= hi = pure ()+ | otherwise = do+ let sumP [] !acc = pure acc+ sumP ((cs, fs) : ps) !acc = do+ x <- VUM.unsafeRead cs c+ when (acc == 0 && x > 0) $+ VUM.unsafeRead fs c >>= VUM.unsafeWrite firstRow c+ sumP ps (acc + x)+ s <- sumP partials 0+ VUM.unsafeWrite totals c s+ go (c + 1)++{- | Wide domains: two-level radix. Rows are bucketed by the top code bits+(cache-resident cursors) as packed @(code, row)@ words; each bucket's slice —+in original row order — yields its exact per-code counts and first rows from an+L1-resident table. No placement pass runs here.+-}+layoutWide ::+ (Int -> Int) ->+ Int ->+ Int ->+ (VU.Vector Int -> (VU.Vector Int, Int)) ->+ IO (Maybe (VU.Vector Int, VU.Vector Int, VU.Vector Int, Int))+layoutWide codeAt n card mkGroups = do+ mPacked <- packBucketed True codeAt n card+ case mPacked of+ Nothing -> pure Nothing+ Just (shift, bucketStart, packed) -> do+ countsM <- VUM.unsafeNew card+ firstRowM <- VUM.unsafeNew card+ overBuckets+ bucketStart+ n+ (countFirstBucket shift card bucketStart packed countsM firstRowM)+ (countFirstBucketPar shift card bucketStart packed countsM firstRowM)+ counts <- VU.unsafeFreeze countsM+ firstRow <- VU.unsafeFreeze firstRowM+ finishLayout codeAt n card mkGroups counts firstRow++{- | Shared tail of 'directLayoutLazy': group mapping, offsets, representative+rows gathered through the code-to-group table, and the parallel @rowToGroup@+pass (one sequential read of the codes, one table lookup each).+-}+finishLayout ::+ (Int -> Int) ->+ Int ->+ Int ->+ (VU.Vector Int -> (VU.Vector Int, Int)) ->+ VU.Vector Int ->+ VU.Vector Int ->+ IO (Maybe (VU.Vector Int, VU.Vector Int, VU.Vector Int, Int))+finishLayout codeAt n card mkGroups counts firstRow = do+ let (!codeToGroup, !nGroups) = mkGroups counts+ offs <- scanOffsets counts codeToGroup nGroups+ repsM <- VUM.unsafeNew nGroups+ let repLoop !c+ | c >= card = pure ()+ | VU.unsafeIndex counts c == 0 = repLoop (c + 1)+ | otherwise = do+ VUM.unsafeWrite+ repsM+ (VU.unsafeIndex codeToGroup c)+ (VU.unsafeIndex firstRow c)+ repLoop (c + 1)+ repLoop 0+ reps <- VU.unsafeFreeze repsM+ rtgM <- VUM.unsafeNew n+ -- A fully occupied ascending domain maps every code to itself; skipping+ -- the per-row random table lookup then leaves one sequential read+write.+ let identity = isIdentityMap codeToGroup+ _ <-+ forkJoin+ [ ( if identity+ then rtgChunkIdentity codeAt rtgM lo hi+ else rtgChunk codeAt codeToGroup rtgM lo hi+ )+ | (lo, hi) <- rowChunks n+ ]+ rtg <- VU.unsafeFreeze rtgM+ pure (Just (rtg, offs, reps, nGroups))++-- | Whether @codeToGroup@ maps every code to itself (fully occupied domain).+isIdentityMap :: VU.Vector Int -> Bool+isIdentityMap m = go 0+ where+ !k = VU.length m+ go !i+ | i >= k = True+ | VU.unsafeIndex m i /= i = False+ | otherwise = go (i + 1)++{- | Exclusive prefix scan of per-group counts (gathered through @codeToGroup@)+into the offsets array of length @nGroups + 1@.+-}+scanOffsets :: VU.Vector Int -> VU.Vector Int -> Int -> IO (VU.Vector Int)+scanOffsets counts codeToGroup nGroups = do+ let !card = VU.length counts+ grpCount <- VUM.new nGroups+ let gather !c+ | c >= card = pure ()+ | otherwise = do+ let !cnt = VU.unsafeIndex counts c+ if cnt == 0+ then gather (c + 1)+ else do+ VUM.unsafeWrite grpCount (VU.unsafeIndex codeToGroup c) cnt+ gather (c + 1)+ gather 0+ offsM <- VUM.new (nGroups + 1)+ let scan !g !acc+ | g >= nGroups = VUM.unsafeWrite offsM nGroups acc+ | otherwise = do+ VUM.unsafeWrite offsM g acc+ c <- VUM.unsafeRead grpCount g+ scan (g + 1) (acc + c)+ scan 0 0+ VU.unsafeFreeze offsM++-- | @rowToGroup@ for one row chunk: remap each row's code through the table.+rtgChunk ::+ (Int -> Int) -> VU.Vector Int -> VUM.IOVector Int -> Int -> Int -> IO ()+rtgChunk codeAt codeToGroup rtgM lo hi = go lo+ where+ go !i+ | i >= hi = pure ()+ | otherwise = do+ let !c = codeAt i+ VUM.unsafeWrite rtgM i (VU.unsafeIndex codeToGroup c)+ go (i + 1)++-- | 'rtgChunk' without the remap lookup (codeToGroup is the identity).+rtgChunkIdentity :: (Int -> Int) -> VUM.IOVector Int -> Int -> Int -> IO ()+rtgChunkIdentity codeAt rtgM lo hi = go lo+ where+ go !i+ | i >= hi = pure ()+ | otherwise = do+ VUM.unsafeWrite rtgM i (codeAt i)+ go (i + 1)++-------------------------------------------------------------------------------+-- Deferred stable placement: valueIndices from rowToGroup+-------------------------------------------------------------------------------++{- | The unique stable counting-sort permutation of @rowToGroup@: rows sorted by+group id, original order within each group — exactly what the eager engines'+placement pass produces. Pure (integer bookkeeping only, deterministic), so it+can sit under a lazy 'DataFrame.Internal.DataFrame.valueIndices' field.++Preconditions (all guaranteed by the grouping paths): @rtg@ has length @n@ with+every value in @[0, nGroups)@, and @offs@ is the group-count prefix array of+length @nGroups + 1@ with @offs[nGroups] == n@.+-}+visFromRowToGroup ::+ Int -> Int -> VU.Vector Int -> VU.Vector Int -> VU.Vector Int+visFromRowToGroup n nGroups offs rtg+ | n <= 0 = VU.empty+ | useTwoLevel n nGroups = unsafePerformIO (visWide n nGroups offs rtg)+ | otherwise = unsafePerformIO (visNarrow n nGroups offs rtg)+{-# NOINLINE visFromRowToGroup #-}++{- | Narrow group domains: per-chunk group histograms prefix-summed (chunk+order) into disjoint cursors seeded from @offs@, then parallel stable placement.+-}+visNarrow :: Int -> Int -> VU.Vector Int -> VU.Vector Int -> IO (VU.Vector Int)+visNarrow n nGroups offs rtg = do+ let chunks = rowChunks n+ hists <-+ forkJoin+ [ do+ acc <- VUM.replicate nGroups (0 :: Int)+ let go !i+ | i >= hi = pure acc+ | otherwise = do+ let !g = VU.unsafeIndex rtg i+ x <- VUM.unsafeRead acc g+ VUM.unsafeWrite acc g (x + 1)+ go (i + 1)+ go lo+ | (lo, hi) <- chunks+ ]+ -- Rewrite each chunk histogram into its write cursor:+ -- cursor_w[g] = offs[g] + Σ_{w'<w} hist_w'[g].+ _ <-+ forkJoin+ [ let seed !g+ | g >= hi = pure ()+ | otherwise = do+ let inner [] !_ = pure ()+ inner (h : hs) !a = do+ t <- VUM.unsafeRead h g+ VUM.unsafeWrite h g a+ inner hs (a + t)+ inner hists (VU.unsafeIndex offs g)+ seed (g + 1)+ in seed lo+ | (lo, hi) <- codeSlices nGroups+ ]+ visM <- VUM.unsafeNew n+ _ <-+ forkJoin+ [ let place !i+ | i >= hi = pure ()+ | otherwise = do+ let !g = VU.unsafeIndex rtg i+ p <- VUM.unsafeRead cursor g+ VUM.unsafeWrite visM p i+ VUM.unsafeWrite cursor g (p + 1)+ place (i + 1)+ in place lo+ | ((lo, hi), cursor) <- zip chunks hists+ ]+ VU.unsafeFreeze visM++{- | Wide group domains: two-level radix. Group ids ascend with buckets, so the+bucket-sorted layout written at @offs@-seeded cursors IS @valueIndices@ — each+bucket writes one contiguous region.+-}+visWide :: Int -> Int -> VU.Vector Int -> VU.Vector Int -> IO (VU.Vector Int)+visWide n nGroups offs rtg = do+ mPacked <- packBucketed False (VU.unsafeIndex rtg) n nGroups+ case mPacked of+ Nothing -> visNarrow n nGroups offs rtg -- unreachable: no validation+ Just (shift, bucketStart, packed) -> do+ visM <- VUM.unsafeNew n+ overBuckets+ bucketStart+ n+ (placeBucketOffs shift nGroups bucketStart offs packed visM)+ (placeBucketOffsPar shift nGroups bucketStart offs packed visM)+ VU.unsafeFreeze visM++{- | Stable placement of one bucket's packed slice at cursors seeded straight+from the group offsets (the bucket's groups own a contiguous @vis@ region).+-}+placeBucketOffs ::+ Int ->+ Int ->+ VU.Vector Int ->+ VU.Vector Int ->+ PackedBuckets ->+ VUM.IOVector Int ->+ Int ->+ IO ()+placeBucketOffs shift card bucketStart offs (PackedBuckets packed) visM b = do+ let !s = VU.unsafeIndex bucketStart b+ !e = VU.unsafeIndex bucketStart (b + 1)+ !base = b `unsafeShiftL` shift+ !range = min (1 `unsafeShiftL` shift) (card - base)+ cursor <- VUM.unsafeNew range+ let initC !j+ | j >= range = pure ()+ | otherwise = do+ VUM.unsafeWrite cursor j (VU.unsafeIndex offs (base + j))+ initC (j + 1)+ initC 0+ let place !pos+ | pos >= e = pure ()+ | otherwise = do+ pc <- VUM.unsafeRead packed pos+ let !j = (pc `unsafeShiftR` packShift) - base+ p <- VUM.unsafeRead cursor j+ VUM.unsafeWrite visM p (pc .&. packRowMask)+ VUM.unsafeWrite cursor j (p + 1)+ place (pos + 1)+ place s++{- | 'placeBucketOffs' for an oversized bucket: per-sub-chunk histograms+prefix-summed (sub-chunks in row order) onto the offset-seeded cursors keep the+placement identical to the serial walk.+-}+placeBucketOffsPar ::+ Int ->+ Int ->+ VU.Vector Int ->+ VU.Vector Int ->+ PackedBuckets ->+ VUM.IOVector Int ->+ Int ->+ IO ()+placeBucketOffsPar shift card bucketStart offs pb visM b = do+ let !s = VU.unsafeIndex bucketStart b+ !e = VU.unsafeIndex bucketStart (b + 1)+ !base = b `unsafeShiftL` shift+ !range = min (1 `unsafeShiftL` shift) (card - base)+ packed = packedVec pb+ subChunks = [(s + lo, s + hi) | (lo, hi) <- splitChunkRange capabilities (e - s)]+ hists <-+ forkJoin+ [subHist pb base range lo hi | (lo, hi) <- subChunks]+ let seed !j+ | j >= range = pure ()+ | otherwise = do+ let inner [] !_ = pure ()+ inner (h : hs) !a = do+ t <- VUM.unsafeRead h j+ VUM.unsafeWrite h j a+ inner hs (a + t)+ inner hists (VU.unsafeIndex offs (base + j))+ seed (j + 1)+ seed 0+ _ <-+ forkJoin+ [ let place !pos+ | pos >= hi = pure ()+ | otherwise = do+ pc <- VUM.unsafeRead packed pos+ let !j = (pc `unsafeShiftR` packShift) - base+ p <- VUM.unsafeRead cursor j+ VUM.unsafeWrite visM p (pc .&. packRowMask)+ VUM.unsafeWrite cursor j (p + 1)+ place (pos + 1)+ in place lo+ | ((lo, hi), cursor) <- zip subChunks hists+ ]+ pure ()++-------------------------------------------------------------------------------+-- Two-level radix plumbing+-------------------------------------------------------------------------------++{- | Above this code-domain size the parallel passes switch to the two-level+radix engine; below it, per-worker direct tables stay cache-resident and are+faster (no bucket store).+-}+twoLevelCardThreshold :: Int+twoLevelCardThreshold = 1024++{- | The two-level engine packs @(code, row)@ into one machine word: row in the+low 'packShift' bits, code above them. Codes are capped at 'directGroupThreshold'+(@2^20@) by every caller, so the packed value stays well within 63 bits; the+guards in 'useTwoLevel' keep the narrow engine for anything larger.+-}+packShift :: Int+packShift = 40++packRowMask :: Int+packRowMask = (1 `unsafeShiftL` packShift) - 1++-- | Bucket-count target of the two-level engine (@2^10@ buckets).+bucketBits :: Int+bucketBits = 10++-- | Use the two-level engine? (Parallel-scale @n@, wide but packable domain.)+useTwoLevel :: Int -> Int -> Bool+useTwoLevel n card =+ shouldPar n+ && card > twoLevelCardThreshold+ && card <= (1 `unsafeShiftL` 22)+ && n <= packRowMask++-- | @ceilLog2 x@: smallest @s@ with @2^s >= x@ (for @x >= 1@).+ceilLog2 :: Int -> Int+ceilLog2 x+ | x <= 1 = 0+ | otherwise = 64 - countLeadingZeros (x - 1)++{- | The bucket store: rows partitioned by the top code bits, each bucket+holding packed @(code, row)@ words in original row order.+-}+newtype PackedBuckets = PackedBuckets (VUM.IOVector Int)++packedVec :: PackedBuckets -> VUM.IOVector Int+packedVec (PackedBuckets v) = v++{- | Partition rows into ~@2^'bucketBits'@ buckets by the top bits of their+code, as packed @(code, row)@ words: per-chunk bucket histograms (validating+every code when asked), prefix-summed in (bucket, chunk) order into disjoint+cursors, then a parallel scatter. Chunks are processed in row order, so every+bucket keeps its rows in ascending original row order. Returns 'Nothing' iff+validation was requested and some code fell outside @[0, card)@.+-}+packBucketed ::+ Bool ->+ (Int -> Int) ->+ Int ->+ Int ->+ IO (Maybe (Int, VU.Vector Int, PackedBuckets))+packBucketed validate codeAt n card = do+ let !shift = max 0 (ceilLog2 card - bucketBits)+ !nBuckets = ((card - 1) `unsafeShiftR` shift) + 1+ chunks = rowChunks n+ parts <-+ forkJoin+ [bucketHist validate codeAt card shift nBuckets lo hi | (lo, hi) <- chunks]+ if not (all snd parts)+ then pure Nothing+ else do+ let cursors = map fst parts+ bucketStartM <- VUM.unsafeNew (nBuckets + 1)+ let seed !b !acc+ | b >= nBuckets = VUM.unsafeWrite bucketStartM nBuckets acc+ | otherwise = do+ VUM.unsafeWrite bucketStartM b acc+ let inner [] !a = pure a+ inner (cur : rest) !a = do+ t <- VUM.unsafeRead cur b+ VUM.unsafeWrite cur b a+ inner rest (a + t)+ acc' <- inner cursors acc+ seed (b + 1) acc'+ seed 0 0+ bucketStart <- VU.unsafeFreeze bucketStartM+ packed <- VUM.unsafeNew n+ _ <-+ forkJoin+ [ scatterPacked codeAt shift cur packed lo hi+ | ((lo, hi), cur) <- zip chunks cursors+ ]+ pure (Just (shift, bucketStart, PackedBuckets packed))++{- | Histogram one row chunk by bucket (top code bits) into a private+@nBuckets@-slot count; with @validate@, report 'False' as soon as any code+escapes @[0, card)@ (the counts are then abandoned).+-}+bucketHist ::+ Bool ->+ (Int -> Int) ->+ Int ->+ Int ->+ Int ->+ Int ->+ Int ->+ IO (VUM.IOVector Int, Bool)+bucketHist validate codeAt card shift nBuckets lo hi = do+ acc <- VUM.replicate nBuckets (0 :: Int)+ let bump !c !i = do+ let !b = c `unsafeShiftR` shift+ x <- VUM.unsafeRead acc b+ VUM.unsafeWrite acc b (x + 1)+ go (i + 1)+ go !i+ | i >= hi = pure True+ | otherwise = do+ let !c = codeAt i+ if validate && (c < 0 || c >= card)+ then pure False+ else bump c i+ ok <- go lo+ pure (acc, ok)++-- | Scatter one row chunk's packed @(code, row)@ words through its bucket cursor.+scatterPacked ::+ (Int -> Int) ->+ Int ->+ VUM.IOVector Int ->+ VUM.IOVector Int ->+ Int ->+ Int ->+ IO ()+scatterPacked codeAt shift cursor packed lo hi = go lo+ where+ go !i+ | i >= hi = pure ()+ | otherwise = do+ let !c = codeAt i+ !b = c `unsafeShiftR` shift+ pos <- VUM.unsafeRead cursor b+ VUM.unsafeWrite packed pos ((c `unsafeShiftL` packShift) + i)+ VUM.unsafeWrite cursor b (pos + 1)+ go (i + 1)++{- | Drive one action per bucket: buckets far above the fair per-worker share+run first through @big@ (internally parallel, one at a time), the rest are+pulled off a shared counter by one worker per capability. Every bucket —+including empty ones — is visited exactly once, so per-bucket passes may rely+on covering their whole output slice.+-}+overBuckets :: VU.Vector Int -> Int -> (Int -> IO ()) -> (Int -> IO ()) -> IO ()+overBuckets bucketStart n small big = do+ let !nBuckets = VU.length bucketStart - 1+ !bigCut = max parThreshold (2 * (n `div` max 1 capabilities))+ size b = VU.unsafeIndex bucketStart (b + 1) - VU.unsafeIndex bucketStart b+ mapM_ big [b | b <- [0 .. nBuckets - 1], size b >= bigCut]+ pooledIndices capabilities nBuckets $ \b ->+ when (size b < bigCut) (small b)++{- | One bucket's exact per-code counts and first rows from its (row-ordered)+packed slice, via an L1-resident table spanning only the bucket's code range.+Writes the bucket's whole slice of @counts@ (zeros included); @firstRow@ only+where the count is nonzero.+-}+countFirstBucket ::+ Int ->+ Int ->+ VU.Vector Int ->+ PackedBuckets ->+ VUM.IOVector Int ->+ VUM.IOVector Int ->+ Int ->+ IO ()+countFirstBucket shift card bucketStart (PackedBuckets packed) countsM firstRowM b = do+ let !s = VU.unsafeIndex bucketStart b+ !e = VU.unsafeIndex bucketStart (b + 1)+ !base = b `unsafeShiftL` shift+ !range = min (1 `unsafeShiftL` shift) (card - base)+ local <- VUM.replicate range (0 :: Int)+ let hist !pos+ | pos >= e = pure ()+ | otherwise = do+ pc <- VUM.unsafeRead packed pos+ let !j = (pc `unsafeShiftR` packShift) - base+ x <- VUM.unsafeRead local j+ when (x == 0) $+ VUM.unsafeWrite firstRowM (base + j) (pc .&. packRowMask)+ VUM.unsafeWrite local j (x + 1)+ hist (pos + 1)+ hist s+ let flush !j+ | j >= range = pure ()+ | otherwise = do+ t <- VUM.unsafeRead local j+ VUM.unsafeWrite countsM (base + j) t+ flush (j + 1)+ flush 0++-- | 'countFirstBucket' for an oversized bucket, chunked across capabilities.+countFirstBucketPar ::+ Int ->+ Int ->+ VU.Vector Int ->+ PackedBuckets ->+ VUM.IOVector Int ->+ VUM.IOVector Int ->+ Int ->+ IO ()+countFirstBucketPar shift card bucketStart pb countsM firstRowM b = do+ let !s = VU.unsafeIndex bucketStart b+ !e = VU.unsafeIndex bucketStart (b + 1)+ !base = b `unsafeShiftL` shift+ !range = min (1 `unsafeShiftL` shift) (card - base)+ packed = packedVec pb+ subChunks = [(s + lo, s + hi) | (lo, hi) <- splitChunkRange capabilities (e - s)]+ parts <-+ forkJoin+ [ do+ local <- VUM.replicate range (0 :: Int)+ firstL <- VUM.unsafeNew range+ let hist !pos+ | pos >= hi = pure (local, firstL)+ | otherwise = do+ pc <- VUM.unsafeRead packed pos+ let !j = (pc `unsafeShiftR` packShift) - base+ x <- VUM.unsafeRead local j+ when (x == 0) $+ VUM.unsafeWrite firstL j (pc .&. packRowMask)+ VUM.unsafeWrite local j (x + 1)+ hist (pos + 1)+ hist lo+ | (lo, hi) <- subChunks+ ]+ -- Merge in sub-chunk (= row) order: totals and global first occurrence.+ let merge !j+ | j >= range = pure ()+ | otherwise = do+ let inner [] !acc = pure acc+ inner ((cs, fs) : ps) !acc = do+ x <- VUM.unsafeRead cs j+ when (acc == 0 && x > 0) $+ VUM.unsafeRead fs j >>= VUM.unsafeWrite firstRowM (base + j)+ inner ps (acc + x)+ t <- inner parts 0+ VUM.unsafeWrite countsM (base + j) t+ merge (j + 1)+ merge 0++-- | Per-sub-chunk histogram of one bucket's packed slice (codes only).+subHist :: PackedBuckets -> Int -> Int -> Int -> Int -> IO (VUM.IOVector Int)+subHist (PackedBuckets packed) base range lo hi = do+ acc <- VUM.replicate range (0 :: Int)+ let go !pos+ | pos >= hi = pure acc+ | otherwise = do+ pc <- VUM.unsafeRead packed pos+ let !j = (pc `unsafeShiftR` packShift) - base+ x <- VUM.unsafeRead acc j+ VUM.unsafeWrite acc j (x + 1)+ go (pos + 1)+ go lo++{- | The ascending-code group order: walk the counts in code order, assigning a+dense group id to each non-empty code (empty codes get no id and no output+group). The single-Int-key path keeps its groups in ascending value order.+-}+ascendingCodeGroups :: VU.Vector Int -> (VU.Vector Int, Int)+ascendingCodeGroups counts = runST $ do+ let !card = VU.length counts+ m <- VUM.new card+ let go !c !next+ | c >= card = pure next+ | VU.unsafeIndex counts c > 0 = do+ VUM.unsafeWrite m c next+ go (c + 1) (next + 1)+ | otherwise = go (c + 1) next+ nGroups <- go 0 0+ frozen <- VU.unsafeFreeze m+ pure (frozen, nGroups)
+ src-internal/DataFrame/Internal/Grouping/Partitioned.hs view
@@ -0,0 +1,406 @@+{-# LANGUAGE BangPatterns #-}+{-# LANGUAGE ScopedTypeVariables #-}+{-# LANGUAGE Strict #-}++{- | Partitioned group-by: rows are counting-sorted into partitions by the top+hash bits, then one task per capability groups its partitions independently,+after which the group numbering is canonicalized to first-appearance order.+Output is bit-for-bit identical to the sequential+'DataFrame.Internal.Grouping.groupBy'.++The name is the mechanism, not the threading: this is a genuinely different+algorithm from the sequential single-hash-table path, not that path with a+fork\/join wrapped around it. (Its sibling+"DataFrame.Internal.Grouping.Direct" is also internally parallel.) Whether to+take this path is 'DataFrame.Internal.Grouping.groupBy''s decision, not this+module's.+-}+module DataFrame.Internal.Grouping.Partitioned (+ parallelAssignGroups,+ rtgFromVisOffs,+ numPartitionsFor,+) where++import Control.Concurrent (getNumCapabilities)+import Control.Monad (forM_, when)+import Data.Bits (countLeadingZeros, unsafeShiftR)+import qualified Data.Vector as V+import qualified Data.Vector.Mutable as VM+import qualified Data.Vector.Unboxed as VU+import qualified Data.Vector.Unboxed.Mutable as VUM+import Data.Word (Word64)+import DataFrame.Internal.Algorithms.Rank.Radix (rankByHash)+import DataFrame.Internal.Control.Concurrent (+ forkJoin,+ forkJoin_,+ parThreshold,+ parallelChunks_,+ pooledIndices,+ )+import DataFrame.Internal.Data.HashTable (+ htInsert,+ newHashTable,+ )+import System.IO.Unsafe (unsafePerformIO)++{- | Sign-preserving unsigned remap: ascending 'Word64' order of @key h@ equals+ascending signed-'Int' order of @h@, so partitioning and sorting on it reproduce+the sequential @compare \`on\` repHash@ ordering exactly.+-}+key :: Int -> Word64+key h = fromIntegral h + 0x8000000000000000+{-# INLINE key #-}++-- | Partition index of a hash: the top @log2 p@ bits of its unsigned key.+partIx :: Int -> Int -> Int+partIx shift h = fromIntegral (key h `unsafeShiftR` shift)+{-# INLINE partIx #-}++{- | Number of partitions: a power of two, at least @4 * caps@ (P >> cores for+skew tolerance), floored at 256 — and grown with the row count (up to+'maxPartitions') so a partition's worst-case hash table (every row a distinct+group: @nextPow2Above (2 * rows/p)@ slots x 3 arrays) stays cache-resident+instead of thrashing DRAM. Partitioning is by the top hash bits and canonical+ranking is ascending unsigned key both across and within partitions, so the+grouping output is bit-for-bit identical at ANY partition count; only the+constant matters for speed.+-}+numPartitionsFor :: Int -> Int -> Int+numPartitionsFor caps n = go 1+ where+ base = max 256 (4 * caps)+ go p+ | p < base = go (p * 2)+ | p < maxPartitions && n > p * partRowTarget = go (p * 2)+ | otherwise = p++-- | Cap on partition count (scatter-pass stream count stays manageable).+maxPartitions :: Int+maxPartitions = 4096++-- | Target rows per partition (~24k rows -> 64k-slot table, ~1.5MB).+partRowTarget :: Int+partRowTarget = 24576++-- | @floor (log2 x)@ for a power-of-two @x@.+intLog2 :: Int -> Int+intLog2 x = 63 - countLeadingZeros x+{-# INLINE intLog2 #-}++{- | Parallel group assignment. @parallelAssignGroups n hashes eqRow@ returns+@(valueIndices, offsets)@ in canonical group order. @eqRow a b@ must report+whether rows @a@ and @b@ share all key columns (null-aware). @rowToGroup@ is+NOT built here any more: gather-style aggregation over huge group counts never+reads it, so callers derive it on demand with 'rtgFromVisOffs'.+-}+parallelAssignGroups ::+ Int ->+ VU.Vector Int ->+ (Int -> Int -> Bool) ->+ IO (VU.Vector Int, VU.Vector Int)+parallelAssignGroups n hashes eqRow = do+ caps <- getNumCapabilities+ let !p = numPartitionsFor caps n+ !shift = 64 - intLog2 p+ (partStart, sortedRows, sortedHash) <- partitionRows n hashes p shift+ localGid <- VUM.new (max 1 n)+ canonBoxes <- VM.replicate p (VU.empty :: VU.Vector Int)+ nLocalGroups <- VUM.replicate p (0 :: Int)+ runPartitions+ caps+ p+ partStart+ sortedRows+ sortedHash+ eqRow+ localGid+ canonBoxes+ nLocalGroups+ (globalBase, canonOf, nGroups) <- canonicalize p canonBoxes nLocalGroups+ assemble n p partStart sortedRows localGid globalBase canonOf nGroups++-------------------------------------------------------------------------------+-- Phase 1: counting sort by partition+-------------------------------------------------------------------------------++{- | Bucket every row index into its partition by a counting sort. Returns the+exclusive prefix-sum @partStart@ (length @p+1@, @partStart[p] == n@), the row+indices laid out partition-by-partition in @sortedRows@, and each sorted+position's hash in @sortedHash@ (same layout) so the grouping loop reads its+hashes sequentially instead of a random @hashes[row]@ per row.++Runs chunked across capabilities: per-chunk partition histograms are prefix+summed (in chunk order) into disjoint per-chunk write cursors, so the scatter+threads never contend and each partition keeps its rows in ascending original+row order — bit-for-bit the sequential counting sort's layout.+-}+partitionRows ::+ Int ->+ VU.Vector Int ->+ Int ->+ Int ->+ IO (VU.Vector Int, VU.Vector Int, VU.Vector Int)+partitionRows n hashes p shift = do+ caps <- getNumCapabilities+ let chunks = rowChunks caps n+ cursors <- forkJoin [histChunk hashes p shift lo hi | (lo, hi) <- chunks]+ -- Exclusive prefix over partitions (outer) and chunks (inner): partStart+ -- from the totals, and each chunk's histogram rewritten into its cursor.+ partStartM <- VUM.new (p + 1)+ let seed !pp !acc+ | pp >= p = VUM.unsafeWrite partStartM p acc+ | otherwise = do+ VUM.unsafeWrite partStartM pp acc+ let inner [] !a = pure a+ inner (cur : rest) !a = do+ t <- VUM.unsafeRead cur pp+ VUM.unsafeWrite cur pp a+ inner rest (a + t)+ acc' <- inner cursors acc+ seed (pp + 1) acc'+ seed 0 0+ sortedM <- VUM.new (max 1 n)+ sortedHashM <- VUM.new (max 1 n)+ forkJoin_+ [ scatterChunk hashes shift cur sortedM sortedHashM lo hi+ | ((lo, hi), cur) <- zip chunks cursors+ ]+ partStart <- VU.unsafeFreeze partStartM+ sortedRows <- VU.unsafeFreeze sortedM+ sortedHash <- VU.unsafeFreeze sortedHashM+ pure (partStart, sortedRows, sortedHash)++-- | Contiguous near-equal row chunks, one per capability; empties dropped.+rowChunks :: Int -> Int -> [(Int, Int)]+rowChunks caps n =+ [ (lo, hi)+ | w <- [0 .. caps - 1]+ , let lo = min n (w * per)+ , let hi = min n (lo + per)+ , lo < hi+ ]+ where+ !per = (n + max 1 caps - 1) `div` max 1 caps++-- | Per-partition counts of one row chunk.+histChunk :: VU.Vector Int -> Int -> Int -> Int -> Int -> IO (VUM.IOVector Int)+histChunk hashes p shift lo hi = do+ acc <- VUM.replicate p (0 :: Int)+ let go !i+ | i >= hi = pure acc+ | otherwise = do+ let !pp = partIx shift (VU.unsafeIndex hashes i)+ c <- VUM.unsafeRead acc pp+ VUM.unsafeWrite acc pp (c + 1)+ go (i + 1)+ go lo++{- | Scatter one row chunk into @sortedM@/@sortedHashM@ through the chunk's+private cursor.+-}+scatterChunk ::+ VU.Vector Int ->+ Int ->+ VUM.IOVector Int ->+ VUM.IOVector Int ->+ VUM.IOVector Int ->+ Int ->+ Int ->+ IO ()+scatterChunk hashes shift cursor sortedM sortedHashM lo hi = go lo+ where+ go !i+ | i >= hi = pure ()+ | otherwise = do+ let !h = VU.unsafeIndex hashes i+ !pp = partIx shift h+ pos <- VUM.unsafeRead cursor pp+ VUM.unsafeWrite sortedM pos i+ VUM.unsafeWrite sortedHashM pos h+ VUM.unsafeWrite cursor pp (pos + 1)+ go (i + 1)++-------------------------------------------------------------------------------+-- Phase 2: per-partition grouping (parallel)+-------------------------------------------------------------------------------++{- | Group each partition with its own hash table, then rank its local groups into+canonical order — all inside the parallel worker. Forks @caps@ workers pulling+partition indices off a shared counter; disjoint keys mean no cross-partition merge.+-}+runPartitions ::+ Int ->+ Int ->+ VU.Vector Int ->+ VU.Vector Int ->+ VU.Vector Int ->+ (Int -> Int -> Bool) ->+ VUM.IOVector Int ->+ VM.IOVector (VU.Vector Int) ->+ VUM.IOVector Int ->+ IO ()+runPartitions caps p partStart sortedRows sortedHash eqRow localGid canonBoxes nLocalGroups =+ pooledIndices caps p groupPartition+ where+ groupPartition !pp = do+ let !s = VU.unsafeIndex partStart pp+ !e = VU.unsafeIndex partStart (pp + 1)+ !sz = e - s+ when (sz > 0) $ do+ ht <- newHashTable sz+ repHashM <- VUM.new sz+ let loop !pos !nextGid+ | pos >= e = pure nextGid+ | otherwise = do+ let !row = VU.unsafeIndex sortedRows pos+ !h = VU.unsafeIndex sortedHash pos+ (gid, isNew) <- htInsert ht eqRow nextGid row h+ VUM.unsafeWrite localGid pos gid+ if isNew+ then do+ VUM.unsafeWrite repHashM nextGid h+ loop (pos + 1) (nextGid + 1)+ else loop (pos + 1) nextGid+ ng <- loop s 0+ VUM.unsafeWrite nLocalGroups pp ng+ canon <- rankByHash (VUM.unsafeRead repHashM) ng+ VM.unsafeWrite canonBoxes pp canon++-------------------------------------------------------------------------------+-- Phase 3: global base ids + assembly+-------------------------------------------------------------------------------++{- | Exclusive prefix sum of the per-partition group counts into @globalBase@+(@globalBase[pp]@ = first global id of partition @pp@). Ranks were computed in+'runPartitions'; prepending the base to each yields the sequential order.+-}+canonicalize ::+ Int ->+ VM.IOVector (VU.Vector Int) ->+ VUM.IOVector Int ->+ IO (VU.Vector Int, V.Vector (VU.Vector Int), Int)+canonicalize p canonBoxes nLocalGroups = do+ globalBaseM <- VUM.new (p + 1)+ let go !pp !base+ | pp >= p = VUM.unsafeWrite globalBaseM p base >> pure base+ | otherwise = do+ VUM.unsafeWrite globalBaseM pp base+ ng <- VUM.unsafeRead nLocalGroups pp+ go (pp + 1) (base + ng)+ total <- go 0 0+ globalBase <- VU.unsafeFreeze globalBaseM+ canonOf <- V.unsafeFreeze canonBoxes+ pure (globalBase, canonOf, total)++{- | Build the final @(valueIndices, offsets)@: the global group id of a+sorted position is @globalBase[pp] + canonOf[pp][localGid]@. @valueIndices@ orders+rows by group, @offsets@ the boundaries. (@rowToGroup@, the per-original-row+inverse, is no longer built here — 'rtgFromVisOffs' derives it on demand, so+aggregations that never read it skip its full random-write pass.)++Each partition owns a disjoint @sortedRows@ range and a disjoint global group-id+range, and its rows are exactly its groups' rows — so its first group's offset is+its own @partStart@ and every pass (group ids, offsets, placement) runs per+partition on parallel workers with no shared writes. @sortedRows@ keeps ascending+original row order inside a partition, so per-group row order matches the+sequential pass exactly.+-}+assemble ::+ Int ->+ Int ->+ VU.Vector Int ->+ VU.Vector Int ->+ VUM.IOVector Int ->+ VU.Vector Int ->+ V.Vector (VU.Vector Int) ->+ Int ->+ IO (VU.Vector Int, VU.Vector Int)+assemble n p partStart sortedRows localGid globalBase canonOf nGroups = do+ caps <- getNumCapabilities+ gidAt <- VUM.new (max 1 n)+ counts <- VUM.new (max 1 nGroups)+ offsM <- VUM.new (nGroups + 1)+ visM <- VUM.new (max 1 n)+ let doPartition !pp = do+ let !s = VU.unsafeIndex partStart pp+ !e = VU.unsafeIndex partStart (pp + 1)+ !base = VU.unsafeIndex globalBase pp+ !gEnd = VU.unsafeIndex globalBase (pp + 1)+ !canon = V.unsafeIndex canonOf pp+ let zero !g+ | g >= gEnd = pure ()+ | otherwise = VUM.unsafeWrite counts g 0 >> zero (g + 1)+ zero base+ -- Pass 1: global group ids and per-group counts.+ let pass1 !pos+ | pos >= e = pure ()+ | otherwise = do+ lg <- VUM.unsafeRead localGid pos+ let !g = base + VU.unsafeIndex canon lg+ VUM.unsafeWrite gidAt pos g+ c <- VUM.unsafeRead counts g+ VUM.unsafeWrite counts g (c + 1)+ pass1 (pos + 1)+ pass1 s+ -- Offsets for our group range (they start at our partStart);+ -- counts becomes the per-group write cursor.+ let offsLoop !g !acc+ | g >= gEnd = pure ()+ | otherwise = do+ VUM.unsafeWrite offsM g acc+ c <- VUM.unsafeRead counts g+ VUM.unsafeWrite counts g acc+ offsLoop (g + 1) (acc + c)+ offsLoop base s+ -- Pass 2: stable placement into valueIndices.+ let pass2 !pos+ | pos >= e = pure ()+ | otherwise = do+ g <- VUM.unsafeRead gidAt pos+ let !row = VU.unsafeIndex sortedRows pos+ c <- VUM.unsafeRead counts g+ VUM.unsafeWrite visM c row+ VUM.unsafeWrite counts g (c + 1)+ pass2 (pos + 1)+ pass2 s+ pooledIndices caps p doPartition+ VUM.unsafeWrite offsM nGroups n+ offs <- VU.unsafeFreeze offsM+ vis <- VU.unsafeFreeze visM+ pure (vis, offs)++{- | Deferred @rowToGroup@ from @(valueIndices, offsets)@:+@rtg[vis[i]] = g@ for every @i@ in group @g@'s range. @vis@ is a permutation,+so any split of the position space writes disjoint slots; each worker binary+searches its first group and then walks group ranges. Values are identical to+the @rowToGroup@ the assembly pass used to build inline. Pure w.r.t. its+immutable inputs, so the 'unsafePerformIO' is safe.+-}+rtgFromVisOffs :: Int -> VU.Vector Int -> VU.Vector Int -> VU.Vector Int+rtgFromVisOffs n vis offs = unsafePerformIO $ do+ let !nGroups = VU.length offs - 1+ rtgM <- VUM.new (max 1 n)+ let+ -- Largest g with offs[g] <= i (offsets are non-decreasing).+ findGroup !i = go2 0 nGroups+ where+ go2 !lo !hi+ | lo >= hi = lo - 1+ | otherwise =+ let !mid = (lo + hi) `div` 2+ in if VU.unsafeIndex offs mid <= i+ then go2 (mid + 1) hi+ else go2 lo mid+ fill !i !hi !g+ | i >= hi = pure ()+ | otherwise = do+ let !g' = advance g+ advance !gg =+ if VU.unsafeIndex offs (gg + 1) <= i+ then advance (gg + 1)+ else gg+ VUM.unsafeWrite rtgM (VU.unsafeIndex vis i) g'+ fill (i + 1) hi g'+ parallelChunks_ parThreshold n (\lo hi -> fill lo hi (findGroup lo))+ VU.unsafeFreeze rtgM+{-# NOINLINE rtgFromVisOffs #-}
− src-internal/DataFrame/Internal/GroupingDirect.hs
@@ -1,234 +0,0 @@-{-# LANGUAGE BangPatterns #-}-{-# LANGUAGE ExplicitNamespaces #-}-{-# LANGUAGE GADTs #-}-{-# LANGUAGE ScopedTypeVariables #-}-{-# LANGUAGE TypeApplications #-}--{- | Low-cardinality direct-indexed grouping fast path: when the key is a single-clean unboxed @Int@ column of small value range, the value itself indexes a dense-accumulator (no hashing/probing). Emits groups in ascending value order.--}-module DataFrame.Internal.GroupingDirect (- directGroupThreshold,- tryDirectGroupColumn,- DirectGrouping (..),-) where--import Control.Concurrent (forkIO, getNumCapabilities)-import Control.Concurrent.MVar (newEmptyMVar, putMVar, takeMVar)-import Control.Exception (SomeException, throwIO, try)-import Data.Type.Equality (TestEquality (..), type (:~:) (Refl))-import qualified Data.Vector.Unboxed as VU-import qualified Data.Vector.Unboxed.Mutable as VUM-import System.IO.Unsafe (unsafePerformIO)-import Type.Reflection (typeRep)--import DataFrame.Internal.Column (Column (..))--{- | Largest key value RANGE (max - min + 1) the direct grouping path accepts. A-@2^20@-slot histogram is 8MB; the low-cardinality questions sit far below it-(id4 range 100, id6 range 1e5). Wider ranges fall back to the hash group-by.--}-directGroupThreshold :: Int-directGroupThreshold = 1048576--{- | The grouping layout the hash path also produces: @rowToGroup@, the-group-sorted @valueIndices@, the @offsets@ prefix array, and the group count.--}-data DirectGrouping = DirectGrouping- { dgRowToGroup :: !(VU.Vector Int)- , dgValueIndices :: !(VU.Vector Int)- , dgOffsets :: !(VU.Vector Int)- , dgNGroups :: !Int- }--capabilities :: Int-capabilities = unsafePerformIO getNumCapabilities-{-# NOINLINE capabilities #-}--parThreshold :: Int-parThreshold = 200000--{- | Take the direct path if the (single) key column is a clean non-null unboxed-@Int@ column with a small value range. Returns 'Nothing' to fall back to the-hash group-by on anything else (boxed/text keys, nullable, wide ranges, empty).--}-tryDirectGroupColumn :: Column -> Maybe DirectGrouping-tryDirectGroupColumn (UnboxedColumn Nothing (v :: VU.Vector a))- | Just Refl <- testEquality (typeRep @a) (typeRep @Int)- , not (VU.null v) =- let (!mn, !mx) = rangeOf v- !range = mx - mn + 1- in if range >= 1 && range <= directGroupThreshold- then Just (directGroup v mn range)- else Nothing-tryDirectGroupColumn _ = Nothing---- | Parallel min/max reduce (order-independent).-rangeOf :: VU.Vector Int -> (Int, Int)-rangeOf v- | not (shouldPar n) = rangeChunk v 0 n- | otherwise = unsafePerformIO $ do- let !caps = capabilities- !per = (n + caps - 1) `div` caps- spawn w = do- var <- newEmptyMVar- let !lo = min n (w * per)- !hi = min n (lo + per)- _ <- forkIO (try (pure $! rangeChunk v lo hi) >>= putMVar var)- pure var- vars <- mapM spawn [0 .. caps - 1]- rs <- mapM takeMVar vars- rs' <- mapM (either (throwIO @SomeException) pure) rs- pure (combineRanges (filter (\(a, _) -> a /= maxBound) rs'))- where- !n = VU.length v-{-# NOINLINE rangeOf #-}--rangeChunk :: VU.Vector Int -> Int -> Int -> (Int, Int)-rangeChunk v lo hi = go lo maxBound minBound- where- go !i !mn !mx- | i >= hi = (mn, mx)- | otherwise =- let !x = VU.unsafeIndex v i- in go (i + 1) (min mn x) (max mx x)--combineRanges :: [(Int, Int)] -> (Int, Int)-combineRanges [] = (0, 0)-combineRanges ((a0, b0) : rest) = foldr (\(a, b) (ma, mb) -> (min ma a, max mb b)) (a0, b0) rest--shouldPar :: Int -> Bool-shouldPar n = n >= parThreshold && capabilities > 1--{- | Build the grouping by counting sort on @value - min@: a (parallel) per-value-histogram, compaction of non-empty values into ascending dense ids, a scan into-offsets, then a stable placement pass building @valueIndices@ and @rowToGroup@.--}-directGroup :: VU.Vector Int -> Int -> Int -> DirectGrouping-directGroup v mn range = unsafePerformIO $ do- let !n = VU.length v- hist <- buildHistogram v mn range n- valToGroup <- VUM.replicate range (-1 :: Int)- grpCount <- VUM.new range- nGroups <- compact hist range valToGroup grpCount- offsM <- VUM.new (nGroups + 1)- cursor <- VUM.new nGroups- scanOffsets grpCount nGroups offsM cursor- rtg <- VUM.new n- vis <- VUM.new n- place v mn n valToGroup cursor rtg vis- frozenRtg <- VU.unsafeFreeze rtg- frozenVis <- VU.unsafeFreeze vis- frozenOffs <- VU.unsafeFreeze offsM- pure (DirectGrouping frozenRtg frozenVis frozenOffs nGroups)-{-# NOINLINE directGroup #-}--{- | Parallel per-value histogram: each worker fills a private @range@-slot-count over its row chunk, then the partials are summed (exact integers, so the-merge order is irrelevant). Sequential single pass below 'parThreshold'.--}-buildHistogram :: VU.Vector Int -> Int -> Int -> Int -> IO (VUM.IOVector Int)-buildHistogram v mn range n- | not (shouldPar n) = histChunk v mn range 0 n- | otherwise = do- let !caps = capabilities- !per = (n + caps - 1) `div` caps- spawn w = do- var <- newEmptyMVar- let !lo = min n (w * per)- !hi = min n (lo + per)- _ <- forkIO (try (histChunk v mn range lo hi) >>= putMVar var)- pure var- vars <- mapM spawn [0 .. caps - 1]- rs <- mapM takeMVar vars- parts <- mapM (either (throwIO @SomeException) pure) rs- case parts of- [] -> VUM.replicate range 0- (p0 : rest) -> do- mapM_ (addInto p0 range) rest- pure p0--histChunk :: VU.Vector Int -> Int -> Int -> Int -> Int -> IO (VUM.IOVector Int)-histChunk v mn range lo hi = do- acc <- VUM.replicate range (0 :: Int)- let go !i- | i >= hi = pure ()- | otherwise = do- let !k = VU.unsafeIndex v i - mn- c <- VUM.unsafeRead acc k- VUM.unsafeWrite acc k (c + 1)- go (i + 1)- go lo- pure acc--addInto :: VUM.IOVector Int -> Int -> VUM.IOVector Int -> IO ()-addInto dst range src = go 0- where- go !k- | k >= range = pure ()- | otherwise = do- a <- VUM.unsafeRead dst k- b <- VUM.unsafeRead src k- VUM.unsafeWrite dst k (a + b)- go (k + 1)--{- | Walk the histogram in ascending value order, assigning a dense group id to-each non-empty value and copying its count into @grpCount@ at that id. Returns-the group count.--}-compact ::- VUM.IOVector Int -> Int -> VUM.IOVector Int -> VUM.IOVector Int -> IO Int-compact hist range valToGroup grpCount = go 0 0- where- go !val !next- | val >= range = pure next- | otherwise = do- c <- VUM.unsafeRead hist val- if c == 0- then go (val + 1) next- else do- VUM.unsafeWrite valToGroup val next- VUM.unsafeWrite grpCount next c- go (val + 1) (next + 1)--{- | Exclusive prefix scan of group counts into @offsM@ (length nGroups+1) and-seed the per-group write @cursor@ at each group's start offset.--}-scanOffsets ::- VUM.IOVector Int -> Int -> VUM.IOVector Int -> VUM.IOVector Int -> IO ()-scanOffsets grpCount nGroups offsM cursor = go 0 0- where- go !g !acc- | g >= nGroups = VUM.unsafeWrite offsM nGroups acc- | otherwise = do- VUM.unsafeWrite offsM g acc- VUM.unsafeWrite cursor g acc- c <- VUM.unsafeRead grpCount g- go (g + 1) (acc + c)--{- | Stable placement pass: for each row in original order, look up its group id-through the value map, write @rowToGroup@, and append the row to its group's run-in @valueIndices@ via the advancing cursor (rows keep original order per group).--}-place ::- VU.Vector Int ->- Int ->- Int ->- VUM.IOVector Int ->- VUM.IOVector Int ->- VUM.IOVector Int ->- VUM.IOVector Int ->- IO ()-place v mn n valToGroup cursor rtg vis = go 0- where- go !i- | i >= n = pure ()- | otherwise = do- let !val = VU.unsafeIndex v i - mn- g <- VUM.unsafeRead valToGroup val- VUM.unsafeWrite rtg i g- pos <- VUM.unsafeRead cursor g- VUM.unsafeWrite vis pos i- VUM.unsafeWrite cursor g (pos + 1)- go (i + 1)
− src-internal/DataFrame/Internal/GroupingPar.hs
@@ -1,308 +0,0 @@-{-# LANGUAGE BangPatterns #-}-{-# LANGUAGE ScopedTypeVariables #-}-{-# LANGUAGE Strict #-}--{- | Parallel partitioned group-by: rows are counting-sorted into partitions by the-top hash bits, then one task per capability groups its partitions independently.-Output is bit-for-bit identical to the sequential 'DataFrame.Internal.Grouping.groupBy'.--}-module DataFrame.Internal.GroupingPar (- parallelAssignGroups,- shouldParallelize,- parThreshold,- numPartitionsFor,-) where--import Control.Concurrent (forkIO, getNumCapabilities)-import Control.Concurrent.MVar (newEmptyMVar, putMVar, takeMVar)-import Control.Exception (SomeException, throwIO, try)-import Control.Monad (forM_, when)-import Data.Bits (countLeadingZeros, unsafeShiftR)-import Data.IORef (atomicModifyIORef', newIORef)-import qualified Data.Vector as V-import qualified Data.Vector.Mutable as VM-import qualified Data.Vector.Unboxed as VU-import qualified Data.Vector.Unboxed.Mutable as VUM-import Data.Word (Word64)-import DataFrame.Internal.HashTable (- htInsert,- newHashTable,- )-import DataFrame.Internal.RadixRank (rankByHash)-import System.IO.Unsafe (unsafePerformIO)--{- | Below this many rows the partition/fork overhead is not worth it; 'groupBy'-uses its sequential 'ST' path instead.--}-parThreshold :: Int-parThreshold = 200000--{- | Whether 'groupBy' should take the parallel path: more than one capability-and at least 'parThreshold' rows.--}-shouldParallelize :: Int -> Bool-shouldParallelize n = n >= parThreshold && capabilities > 1-{-# NOINLINE shouldParallelize #-}--capabilities :: Int-capabilities = unsafePerformIO getNumCapabilities-{-# NOINLINE capabilities #-}--{- | Sign-preserving unsigned remap: ascending 'Word64' order of @key h@ equals-ascending signed-'Int' order of @h@, so partitioning and sorting on it reproduce-the sequential @compare \`on\` repHash@ ordering exactly.--}-key :: Int -> Word64-key h = fromIntegral h + 0x8000000000000000-{-# INLINE key #-}---- | Partition index of a hash: the top @log2 p@ bits of its unsigned key.-partIx :: Int -> Int -> Int-partIx shift h = fromIntegral (key h `unsafeShiftR` shift)-{-# INLINE partIx #-}--{- | Number of partitions: a power of two, at least @4 * caps@ (P >> cores for-skew tolerance), floored at 256.--}-numPartitionsFor :: Int -> Int-numPartitionsFor caps = go 1- where- target = max 256 (4 * caps)- go p- | p >= target = p- | otherwise = go (p * 2)---- | @floor (log2 x)@ for a power-of-two @x@.-intLog2 :: Int -> Int-intLog2 x = 63 - countLeadingZeros x-{-# INLINE intLog2 #-}--{- | Parallel group assignment. @parallelAssignGroups n hashes eqRow@ returns-@(rowToGroup, valueIndices, offsets)@ in canonical group order. @eqRow a b@ must-report whether rows @a@ and @b@ share all key columns (null-aware).--}-parallelAssignGroups ::- Int ->- VU.Vector Int ->- (Int -> Int -> Bool) ->- IO (VU.Vector Int, VU.Vector Int, VU.Vector Int)-parallelAssignGroups n hashes eqRow = do- caps <- getNumCapabilities- let !p = numPartitionsFor caps- !shift = 64 - intLog2 p- (partStart, sortedRows) <- partitionRows n hashes p shift- localGid <- VUM.new (max 1 n)- canonBoxes <- VM.replicate p (VU.empty :: VU.Vector Int)- nLocalGroups <- VUM.replicate p (0 :: Int)- runPartitions- caps- p- partStart- sortedRows- hashes- eqRow- localGid- canonBoxes- nLocalGroups- (globalBase, canonOf, nGroups) <- canonicalize p canonBoxes nLocalGroups- assemble n p partStart sortedRows localGid globalBase canonOf nGroups------------------------------------------------------------------------------------ Phase 1: counting sort by partition----------------------------------------------------------------------------------{- | Bucket every row index into its partition by a counting sort. Returns the-exclusive prefix-sum @partStart@ (length @p+1@, @partStart[p] == n@) and the row-indices laid out partition-by-partition in @sortedRows@.--}-partitionRows ::- Int -> VU.Vector Int -> Int -> Int -> IO (VU.Vector Int, VU.Vector Int)-partitionRows n hashes p shift = do- counts <- VUM.replicate (p + 1) (0 :: Int)- let countLoop !i- | i >= n = pure ()- | otherwise = do- let !pp = partIx shift (VU.unsafeIndex hashes i)- c <- VUM.unsafeRead counts pp- VUM.unsafeWrite counts pp (c + 1)- countLoop (i + 1)- countLoop 0- partStartM <- VUM.new (p + 1)- let scan !k !acc- | k > p = pure ()- | otherwise = do- VUM.unsafeWrite partStartM k acc- c <- if k < p then VUM.unsafeRead counts k else pure 0- scan (k + 1) (acc + c)- scan 0 0- cursor <- VUM.new p- forM_ [0 .. p - 1] $ \k -> VUM.unsafeRead partStartM k >>= VUM.unsafeWrite cursor k- sortedM <- VUM.new (max 1 n)- let place !i- | i >= n = pure ()- | otherwise = do- let !pp = partIx shift (VU.unsafeIndex hashes i)- pos <- VUM.unsafeRead cursor pp- VUM.unsafeWrite sortedM pos i- VUM.unsafeWrite cursor pp (pos + 1)- place (i + 1)- place 0- partStart <- VU.unsafeFreeze partStartM- sortedRows <- VU.unsafeFreeze sortedM- pure (partStart, sortedRows)------------------------------------------------------------------------------------ Phase 2: per-partition grouping (parallel)----------------------------------------------------------------------------------{- | Group each partition with its own hash table, then rank its local groups into-canonical order — all inside the parallel worker. Forks @caps@ workers pulling-partition indices off a shared counter; disjoint keys mean no cross-partition merge.--}-runPartitions ::- Int ->- Int ->- VU.Vector Int ->- VU.Vector Int ->- VU.Vector Int ->- (Int -> Int -> Bool) ->- VUM.IOVector Int ->- VM.IOVector (VU.Vector Int) ->- VUM.IOVector Int ->- IO ()-runPartitions caps p partStart sortedRows hashes eqRow localGid canonBoxes nLocalGroups = do- next <- newIORef 0- let groupPartition !pp = do- let !s = VU.unsafeIndex partStart pp- !e = VU.unsafeIndex partStart (pp + 1)- !sz = e - s- when (sz > 0) $ do- ht <- newHashTable sz- repHashM <- VUM.new sz- let loop !pos !nextGid- | pos >= e = pure nextGid- | otherwise = do- let !row = VU.unsafeIndex sortedRows pos- !h = VU.unsafeIndex hashes row- (gid, isNew) <- htInsert ht eqRow nextGid row h- VUM.unsafeWrite localGid pos gid- if isNew- then do- VUM.unsafeWrite repHashM nextGid h- loop (pos + 1) (nextGid + 1)- else loop (pos + 1) nextGid- ng <- loop s 0- VUM.unsafeWrite nLocalGroups pp ng- canon <- rankByHash (VUM.unsafeRead repHashM) ng- VM.unsafeWrite canonBoxes pp canon- worker = do- i <- atomicModifyIORef' next (\j -> (j + 1, j))- when (i < p) $ groupPartition i >> worker- forkJoin_ (replicate caps worker)------------------------------------------------------------------------------------ Phase 3: global base ids + assembly----------------------------------------------------------------------------------{- | Exclusive prefix sum of the per-partition group counts into @globalBase@-(@globalBase[pp]@ = first global id of partition @pp@). Ranks were computed in-'runPartitions'; prepending the base to each yields the sequential order.--}-canonicalize ::- Int ->- VM.IOVector (VU.Vector Int) ->- VUM.IOVector Int ->- IO (VU.Vector Int, V.Vector (VU.Vector Int), Int)-canonicalize p canonBoxes nLocalGroups = do- globalBaseM <- VUM.new (p + 1)- let go !pp !base- | pp >= p = VUM.unsafeWrite globalBaseM p base >> pure base- | otherwise = do- VUM.unsafeWrite globalBaseM pp base- ng <- VUM.unsafeRead nLocalGroups pp- go (pp + 1) (base + ng)- total <- go 0 0- globalBase <- VU.unsafeFreeze globalBaseM- canonOf <- V.unsafeFreeze canonBoxes- pure (globalBase, canonOf, total)--{- | Build the final @(rowToGroup, valueIndices, offsets)@: the global group id of a-sorted position is @globalBase[pp] + canonOf[pp][localGid]@. @valueIndices@ orders-rows by group, @offsets@ the boundaries, @rowToGroup@ the inverse per original row.--}-assemble ::- Int ->- Int ->- VU.Vector Int ->- VU.Vector Int ->- VUM.IOVector Int ->- VU.Vector Int ->- V.Vector (VU.Vector Int) ->- Int ->- IO (VU.Vector Int, VU.Vector Int, VU.Vector Int)-assemble n p partStart sortedRows localGid globalBase canonOf nGroups = do- rtgM <- VUM.new (max 1 n)- counts <- VUM.replicate (nGroups + 1) (0 :: Int)- gidAt <- VUM.new (max 1 n)- let scanPos !pp- | pp >= p = pure ()- | otherwise = do- let !s = VU.unsafeIndex partStart pp- !e = VU.unsafeIndex partStart (pp + 1)- !base = VU.unsafeIndex globalBase pp- !canon = V.unsafeIndex canonOf pp- let inner !pos- | pos >= e = pure ()- | otherwise = do- lg <- VUM.unsafeRead localGid pos- let !g = base + VU.unsafeIndex canon lg- !row = VU.unsafeIndex sortedRows pos- VUM.unsafeWrite gidAt pos g- VUM.unsafeWrite rtgM row g- c <- VUM.unsafeRead counts g- VUM.unsafeWrite counts g (c + 1)- inner (pos + 1)- inner s- scanPos (pp + 1)- scanPos 0- offsM <- VUM.new (nGroups + 1)- let scan !k !acc- | k > nGroups = pure ()- | otherwise = do- VUM.unsafeWrite offsM k acc- c <- if k < nGroups then VUM.unsafeRead counts k else pure 0- scan (k + 1) (acc + c)- scan 0 0- cursor <- VUM.new (max 1 nGroups)- forM_ [0 .. nGroups - 1] $ \k -> VUM.unsafeRead offsM k >>= VUM.unsafeWrite cursor k- visM <- VUM.new (max 1 n)- let placeVis !pos- | pos >= n = pure ()- | otherwise = do- g <- VUM.unsafeRead gidAt pos- let !row = VU.unsafeIndex sortedRows pos- c <- VUM.unsafeRead cursor g- VUM.unsafeWrite visM c row- VUM.unsafeWrite cursor g (c + 1)- placeVis (pos + 1)- placeVis 0- rtg <- VU.unsafeFreeze rtgM- offs <- VU.unsafeFreeze offsM- vis <- VU.unsafeFreeze visM- pure (rtg, vis, offs)------------------------------------------------------------------------------------ Thread fan-out (plain forkIO + MVar join, no sparks)------------------------------------------------------------------------------------ | Run each action on its own thread; rethrow the first failure (in order).-forkJoin_ :: [IO ()] -> IO ()-forkJoin_ actions = do- vars <- mapM spawn actions- results <- mapM takeMVar vars- mapM_ (either (throwIO :: SomeException -> IO ()) pure) results- where- spawn act = do- var <- newEmptyMVar- _ <- forkIO (try act >>= putMVar var)- pure var
− src-internal/DataFrame/Internal/Hash.hs
@@ -1,113 +0,0 @@-{-# LANGUAGE BangPatterns #-}-{-# LANGUAGE CPP #-}-{-# LANGUAGE MagicHash #-}--{- | A poor-man's hash used by 'DataFrame.Internal.Grouping' to bucket rows-without depending on @hashable@. Each value is folded into an 'Int' with an-FxHash-style step (rotate, xor, multiply); small and not cryptographic.--}-module DataFrame.Internal.Hash (- fnvOffset,- nullSalt,- mixInt,- mixDouble,- mixBool,- mixChar,- mixText,- mixBytes,- mixShow,-) where--import Data.Bits (rotateL, unsafeShiftL, unsafeShiftR, xor)-import Data.Char (ord)-import qualified Data.Text as T-import qualified Data.Text.Array as A-#if MIN_VERSION_text(2,1,0)-import Data.Array.Byte (ByteArray (ByteArray))-#else-import Data.Text.Array (Array (ByteArray))-#endif-import Data.Text.Internal (Text (Text))-import GHC.Exts (Int (I#), indexWord8Array#, indexWord8ArrayAsWord64#)-import GHC.Word (Word64 (W64#), Word8 (W8#))--{- | FNV-1a 64-bit offset basis (used as the initial accumulator).-The literal is unsigned and exceeds 'Int' range, so we round-trip through-'Word64' to get the well-defined two's-complement bit pattern.--}-fnvOffset :: Int-fnvOffset = fromIntegral (0xcbf29ce484222325 :: Word64)---- | FNV-1a 64-bit prime.-fnvPrime :: Int-fnvPrime = 0x00000100000001b3--{- | Sentinel mixed in for a /null/ slot, so @Nothing@ does not hash the same as-a present value with equal bits (e.g. @Just 0@). A fixed distinctive constant-keeps null hashing deterministic; a real value equal to it collides only rarely.--}-nullSalt :: Int-nullSalt = fromIntegral (0x9E3779B97F4A7C15 :: Word64)--{- | Mix an 'Int' into the accumulator with an FxHash-style step. The rotate-diffuses each value's bits before the next is folded in, avoiding the structured-collisions a plain xor-then-multiply produces on small/adjacent group keys.--}-mixInt :: Int -> Int -> Int-mixInt acc x = (rotateL acc 13 `xor` x) * fnvPrime-{-# INLINE mixInt #-}--{- | Mix a 'Double' into the accumulator. Loses sub-millisecond precision-but matches the bucketing the old hashable-based code used.--}-mixDouble :: Int -> Double -> Int-mixDouble acc d = mixInt acc (floor (d * 1000))-{-# INLINE mixDouble #-}--mixBool :: Int -> Bool -> Int-mixBool acc b = mixInt acc (if b then 1 else 0)-{-# INLINE mixBool #-}--mixChar :: Int -> Char -> Int-mixChar acc = mixInt acc . ord-{-# INLINE mixChar #-}--{- | Mix a 'T.Text' value into the accumulator over its raw UTF-8 bytes, eight at-a time. Reading a whole 'Word64' per step cuts the multiply count ~8x on long-keys while staying collision-equivalent (UTF-8 is injective).--}-mixText :: Int -> T.Text -> Int-mixText !acc (Text arr off len) = mixBytes acc arr off len-{-# INLINE mixText #-}--{- | Mix a raw UTF-8 byte slice @[off, off+len)@ of a 'Data.Text.Array.Array'-into the accumulator, eight bytes at a time. The shared kernel behind-'mixText' and the packed-text hash path, so the two never drift.--}-mixBytes :: Int -> A.Array -> Int -> Int -> Int-mixBytes !acc arr off len = goBytes (goWords acc off) wordsEnd- where- !(ByteArray ba) = arr- !nWords = len `unsafeShiftR` 3- !wordsEnd = off + (nWords `unsafeShiftL` 3)- !end = off + len- goWords !h !i- | i >= wordsEnd = h- | otherwise =- let !(I# i#) = i- !w = fromIntegral (W64# (indexWord8ArrayAsWord64# ba i#)) :: Int- in goWords (mixInt h w) (i + 8)- goBytes !h !i- | i >= end = h- | otherwise =- let !(I# i#) = i- !b = fromIntegral (W8# (indexWord8Array# ba i#)) :: Int- in goBytes (mixInt h b) (i + 1)-{-# INLINE mixBytes #-}--{- | Fallback for arbitrary 'Show'-able values. Slower but covers types-without a dedicated combinator (e.g. 'Day', 'UTCTime').--}-mixShow :: (Show a) => Int -> a -> Int-mixShow acc = mixText acc . T.pack . show-{-# INLINE mixShow #-}
− src-internal/DataFrame/Internal/HashTable.hs
@@ -1,94 +0,0 @@-{-# LANGUAGE BangPatterns #-}-{-# LANGUAGE RankNTypes #-}-{-# LANGUAGE ScopedTypeVariables #-}--{- | A flat, unboxed, open-addressing (linear-probe) hash table mapping a row's-key-hash to a dense group id, re-verifying the real key on every hash hit to-reject collisions. Runs in any 'PrimMonad' ('ST' for grouping, 'IO' per worker).--}-module DataFrame.Internal.HashTable (- HashTable (..),- newHashTable,- htInsert,- nextPow2Above,-) where--import Control.Monad.Primitive (PrimMonad, PrimState)-import Data.Bits ((.&.))-import qualified Data.Vector.Unboxed.Mutable as VUM--{- | An open-addressing linear-probe table. @htMask@ is @capacity - 1@ (capacity-is a power of two) and maps a hash to its home slot.--}-data HashTable s = HashTable- { htHash :: !(VUM.MVector s Int)- , htGroup :: !(VUM.MVector s Int)- , htRep :: !(VUM.MVector s Int)- , htMask :: !Int- }--{- | Smallest power of two strictly greater than @n@, at least 2. Sizes the-table so the load factor stays below ~0.5 even when every row is a distinct-group.--}-nextPow2Above :: Int -> Int-nextPow2Above n = go 2- where- go !p- | p > n = p- | otherwise = go (p * 2)-{-# INLINE nextPow2Above #-}--{- | Allocate an empty table able to hold up to @n@ distinct groups while-keeping the load factor under ~0.5 (capacity @= nextPow2Above (2*n)@). All-group slots start empty (@-1@).--}-newHashTable :: (PrimMonad m) => Int -> m (HashTable (PrimState m))-newHashTable n = do- let !cap = nextPow2Above (2 * max 1 n)- h <- VUM.unsafeNew cap- g <- VUM.replicate cap (-1)- r <- VUM.unsafeNew cap- pure (HashTable h g r (cap - 1))-{-# INLINE newHashTable #-}--{- | Look up @row@ (with precomputed @hash@) and return its dense group id: an-empty slot starts a new group via @nextGroup@, a stored-hash match is re-verified-with @eqRow@ before reuse. The 'Bool' is 'True' when a new group was created.--}-htInsert ::- (PrimMonad m) =>- HashTable (PrimState m) ->- -- | @eqRow a b@: do rows @a@ and @b@ have equal key columns?- (Int -> Int -> Bool) ->- -- | Next dense group id to assign if this row starts a new group.- Int ->- -- | Row index being inserted.- Int ->- -- | Precomputed hash of the row's key.- Int ->- m (Int, Bool)-htInsert ht eqRow nextGroup row hash = go (hash .&. mask)- where- !mask = htMask ht- !hs = htHash ht- !gs = htGroup ht- !rs = htRep ht- go !slot = do- g <- VUM.unsafeRead gs slot- if g < 0- then do- VUM.unsafeWrite hs slot hash- VUM.unsafeWrite gs slot nextGroup- VUM.unsafeWrite rs slot row- pure (nextGroup, True)- else do- h <- VUM.unsafeRead hs slot- if h == hash- then do- rep <- VUM.unsafeRead rs slot- if eqRow rep row- then pure (g, False)- else go ((slot + 1) .&. mask)- else go ((slot + 1) .&. mask)-{-# INLINE htInsert #-}
src-internal/DataFrame/Internal/Interpreter.hs view
@@ -35,10 +35,10 @@ import qualified Data.Vector.Unboxed.Mutable as VUM import DataFrame.Errors import DataFrame.Internal.Column+import DataFrame.Internal.Column.Bitmap import DataFrame.Internal.DataFrame import DataFrame.Internal.Expression import qualified DataFrame.Internal.Grouping as G-import DataFrame.Internal.Types import Type.Reflection ( Typeable, typeRep,@@ -224,6 +224,10 @@ {-# SPECIALIZE mapColumn :: (Int -> Int) -> Column -> Either DataFrameException Column #-}+-- toDouble on an Int column (hot path for derived arithmetic)+{-# SPECIALIZE mapColumn ::+ (Int -> Double) -> Column -> Either DataFrameException Column+ #-} -- zipWithColumns: binary ops {-# SPECIALIZE zipWithColumns ::@@ -477,20 +481,27 @@ V.generate (VU.length indices) ((vec `V.unsafeIndex`) . (indices `VU.unsafeIndex`))+ !sortedBm = permuteBitmap bm in V.generate nGroups $ \i -> BoxedColumn- (fmap (bitmapSlice (start i) (len i)) bm)+ (fmap (bitmapSlice (start i) (len i)) sortedBm) (V.unsafeSlice (start i) (len i) sorted) UnboxedColumn bm vec -> let !sorted = VU.unsafeBackpermute vec indices+ !sortedBm = permuteBitmap bm in V.generate nGroups $ \i -> UnboxedColumn- (fmap (bitmapSlice (start i) (len i)) bm)+ (fmap (bitmapSlice (start i) (len i)) sortedBm) (VU.unsafeSlice (start i) (len i) sorted) where !nGroups = VU.length os - 1 start i = os `VU.unsafeIndex` i len i = os `VU.unsafeIndex` (i + 1) - start i+ permuteBitmap = fmap $ \bm ->+ buildBitmapFromValid $+ VU.map+ (\r -> if bitmapTestBit bm r then 1 else 0)+ indices {-# INLINE sliceGroups #-} numGroups :: GroupedDataFrame -> Int@@ -831,7 +842,7 @@ Group <$> V.mapM (promoteColumnWith onResult) gs eval ctx expr@(Unary op (inner :: Expr b)) = addContext expr $ do v <- eval @b ctx inner- liftValue (unaryFn op) v+ liftValue (fastUnaryFn @b @a (unaryName op) (unaryFn op)) v eval ctx expr@(Binary op (left :: Expr c) (right :: Expr b)) = addContext expr $ do l <- eval @c ctx left@@ -851,7 +862,7 @@ Flat groupCol -> Right (Flat (atIndicesStable (rowToGroup gdf) groupCol)) Group groupCols -> do- sorted <- V.fold1M' concatColumns groupCols+ sorted <- V.fold1M' mappendColumns groupCols let inv = invertPermutation (valueIndices gdf) Right (Flat (atIndicesStable inv sorted)) eval (GroupCtx _) expr@(Over _ _) =@@ -981,6 +992,37 @@ Group gs -> Flat . fromVector <$> V.mapM (foldl1Column @a f) gs++{- | The op's element function, with a fast path for @toDouble@ (matched by+'unaryName', like the @toDouble@ peeling in "DataFrame.Internal.Simplify").+The closure captured at 'Expr'-construction time is @realToFrac@, which at an+integral source type without a fired rewrite rule lowers to+@fromRational . toRational@ — a 'Rational' allocation plus 'fromRat' per+element. 'fromIntegral' at the concrete type is the same correctly-rounded+conversion, so the swap is bit-identical; only the constant factor changes.+Non-integral sources and other ops keep the stored function.+-}+fastUnaryFn ::+ forall b a.+ (Columnable b, Columnable a) =>+ T.Text -> (b -> a) -> (b -> a)+fastUnaryFn name f+ | name == "toDouble"+ , Just Refl <- testEquality (typeRep @a) (typeRep @Double) =+ integralToDouble @b f+ | otherwise = f++integralToDouble :: forall b. (Columnable b) => (b -> Double) -> b -> Double+integralToDouble f+ | Just Refl <- testEquality rb (typeRep @Int) = fromIntegral+ | Just Refl <- testEquality rb (typeRep @Int8) = fromIntegral+ | Just Refl <- testEquality rb (typeRep @Int16) = fromIntegral+ | Just Refl <- testEquality rb (typeRep @Int32) = fromIntegral+ | Just Refl <- testEquality rb (typeRep @Int64) = fromIntegral+ | Just Refl <- testEquality rb (typeRep @Word) = fromIntegral+ | otherwise = f+ where+ rb = typeRep @b broadcastFold :: forall acc b.
− src-internal/DataFrame/Internal/Nullable.hs
@@ -1,467 +0,0 @@-{-# LANGUAGE AllowAmbiguousTypes #-}-{-# LANGUAGE FlexibleContexts #-}-{-# LANGUAGE FlexibleInstances #-}-{-# LANGUAGE FunctionalDependencies #-}-{-# LANGUAGE ScopedTypeVariables #-}-{-# LANGUAGE TypeApplications #-}-{-# LANGUAGE TypeFamilies #-}-{-# LANGUAGE TypeOperators #-}-{-# LANGUAGE UndecidableInstances #-}-{-# LANGUAGE UndecidableSuperClasses #-}--{- | Nullable-aware arithmetic and comparison operators ('.+', '.==', …) that work-transparently across nullable (@Maybe a@) and non-nullable (@a@) operands.-Functional dependencies infer the result type without annotations.--@--- Mixing nullable and non-nullable columns:-F.col \@Int \"x\" '.+' F.col \@(Maybe Int) \"y\" -- :: Expr (Maybe Int)---- Both non-nullable (existing behaviour preserved):-F.col \@Int \"x\" '.+' F.col \@Int \"y\" -- :: Expr Int---- Comparison with three-valued logic:-F.col \@(Maybe Int) \"x\" '.==' F.col \@Int \"y\" -- :: Expr (Maybe Bool)-@--}-module DataFrame.Internal.Nullable (- -- * Type family- BaseType,-- -- * Arithmetic class- NullableArithOp (..),-- -- * Comparison class- NullableCmpOp (..),-- -- * Generalized nullable lift classes- NullLift1Op (..),- NullLift2Op (..),-- -- * Result-type type families (drive inference in nullLift / nullLift2)- NullLift1Result,- NullLift2Result,-- -- * Result-type type family for comparison operators- NullCmpResult,-- -- * Numeric widening- NumericWidenOp (..),- widenArithOp,- widenCmpOp,- WidenResult,-- -- * Division widening (integral × integral → Double)- DivWidenOp (..),- divArithOp,- WidenResultDiv,-) where--import Data.Int (Int32, Int64)-import DataFrame.Internal.Column (Columnable)-import DataFrame.Internal.Types (Promote, PromoteDiv)--{- | Strip one layer of 'Maybe'.--@-BaseType (Maybe a) = a-BaseType a = a -- for any non-Maybe type-@--}-type family BaseType a where- BaseType (Maybe a) = a- BaseType a = a--{- | Arithmetic binary operations that work over nullable and non-nullable operand-types. The functional dependency @a b -> c@ infers the result; the 'OVERLAPPABLE'-non-nullable instance yields to the specific @(Maybe a, Maybe a)@ one.--}-class- ( Columnable a- , Columnable b- , Columnable c- ) =>- NullableArithOp a b c- | a b -> c- where- {- | Lift an arithmetic function over the inner (non-Maybe) values.- 'Nothing' short-circuits: any 'Nothing' operand produces 'Nothing'.- -}- nullArithOp ::- (BaseType a -> BaseType a -> BaseType a) ->- a ->- b ->- c--{- | Compute the result type of a nullable comparison.--@-NullCmpResult (Maybe a) b = Maybe Bool-NullCmpResult a (Maybe b) = Maybe Bool -- when a is apart from Maybe-NullCmpResult a b = Bool-@--Used by the comparison operators ('.==', '.<', etc.) so GHC infers the-return type without an explicit annotation.--}-type family NullCmpResult a b where- NullCmpResult (Maybe a) b = Maybe Bool- NullCmpResult a (Maybe b) = Maybe Bool- NullCmpResult a b = Bool--{- | Comparison binary operations over nullable and non-nullable operands. No-functional dependency on @e@; overlapping/overlappable instance pragmas pick the-unique most-specific instance from the concrete operand types.--}-class- ( Columnable a- , Columnable b- , Columnable e- ) =>- NullableCmpOp a b e- where- {- | Lift a comparison function over the inner values (three-valued logic).- Returns 'Nothing' when either operand is 'Nothing'.- -}- nullCmpOp ::- (BaseType a -> BaseType a -> Bool) ->- a ->- b ->- e--{- | Non-nullable × Non-nullable: apply directly, no wrapping.-Arithmetic result is @a@; comparison result is @Bool@.--}-instance- {-# OVERLAPPABLE #-}- (Columnable a, a ~ BaseType a) =>- NullableArithOp a a a- where- nullArithOp f = f--instance- {-# OVERLAPPABLE #-}- (Columnable a, Columnable Bool, a ~ BaseType a) =>- NullableCmpOp a a Bool- where- nullCmpOp f = f---- | Nullable × Non-nullable: 'Nothing' short-circuits.-instance- (Columnable a, Columnable (Maybe a)) =>- NullableArithOp (Maybe a) a (Maybe a)- where- nullArithOp _f Nothing _ = Nothing- nullArithOp f (Just x) y = Just (f x y)--instance- (Columnable a, Columnable (Maybe a), Columnable (Maybe Bool)) =>- NullableCmpOp (Maybe a) a (Maybe Bool)- where- nullCmpOp _f Nothing _ = Nothing- nullCmpOp f (Just x) y = Just (f x y)---- | Non-nullable × Nullable: 'Nothing' short-circuits.-instance- ( Columnable a- , Columnable (Maybe a)- , a ~ BaseType a- ) =>- NullableArithOp a (Maybe a) (Maybe a)- where- nullArithOp _f _ Nothing = Nothing- nullArithOp f x (Just y) = Just (f x y)--instance- ( Columnable a- , Columnable (Maybe a)- , Columnable (Maybe Bool)- , a ~ BaseType a- ) =>- NullableCmpOp a (Maybe a) (Maybe Bool)- where- nullCmpOp _f _ Nothing = Nothing- nullCmpOp f x (Just y) = Just (f x y)---- | Nullable × Nullable: either 'Nothing' short-circuits.-instance- {-# OVERLAPPING #-}- (Columnable a, Columnable (Maybe a)) =>- NullableArithOp (Maybe a) (Maybe a) (Maybe a)- where- nullArithOp _f Nothing _ = Nothing- nullArithOp _f _ Nothing = Nothing- nullArithOp f (Just x) (Just y) = Just (f x y)--instance- {-# OVERLAPPING #-}- (Columnable a, Columnable (Maybe a), Columnable (Maybe Bool)) =>- NullableCmpOp (Maybe a) (Maybe a) (Maybe Bool)- where- nullCmpOp _f Nothing _ = Nothing- nullCmpOp _f _ Nothing = Nothing- nullCmpOp f (Just x) (Just y) = Just (f x y)---- ------------------------------------------------------------------------------ Generalized nullable lift (unary)--- -----------------------------------------------------------------------------{- | Lift a unary function over a column expression, propagating 'Nothing' (applied-directly when non-nullable, under 'Just' when @a = Maybe x@). Use via-'DataFrame.Functions.nullLift'.--}--{- | Compute the result type of a nullable unary lift.--@-NullLift1Result (Maybe a) r = Maybe r-NullLift1Result a r = r -- for any non-Maybe a-@--Used by 'DataFrame.Functions.nullLift' so GHC can infer the return type-without an explicit annotation.--}-type family NullLift1Result a r where- NullLift1Result (Maybe a) r = Maybe r- NullLift1Result a r = r--class- ( Columnable a- , Columnable r- , Columnable c- ) =>- NullLift1Op a r c- where- applyNull1 :: (BaseType a -> r) -> a -> c---- | Non-nullable: apply directly.-instance- {-# OVERLAPPABLE #-}- (Columnable a, Columnable r, a ~ BaseType a) =>- NullLift1Op a r r- where- applyNull1 f = f---- | Nullable: propagate 'Nothing'.-instance- {-# OVERLAPPING #-}- (Columnable a, Columnable r, Columnable (Maybe r)) =>- NullLift1Op (Maybe a) r (Maybe r)- where- applyNull1 _ Nothing = Nothing- applyNull1 f (Just x) = Just (f x)---- ------------------------------------------------------------------------------ Generalized nullable lift (binary)--- -----------------------------------------------------------------------------{- | Lift a binary function over two column expressions, propagating 'Nothing': the-result is @Maybe r@ if either operand is nullable, else @r@. Use via-'DataFrame.Functions.nullLift2'.--}--{- | Compute the result type of a nullable binary lift.--@-NullLift2Result (Maybe a) b r = Maybe r-NullLift2Result a (Maybe b) r = Maybe r -- when a is apart from Maybe-NullLift2Result a b r = r-@--Used by 'DataFrame.Functions.nullLift2' so GHC can infer the return type.--}-type family NullLift2Result a b r where- NullLift2Result (Maybe a) b r = Maybe r- NullLift2Result a (Maybe b) r = Maybe r- NullLift2Result a b r = r--class- ( Columnable a- , Columnable b- , Columnable r- , Columnable c- ) =>- NullLift2Op a b r c- where- applyNull2 :: (BaseType a -> BaseType b -> r) -> a -> b -> c---- | Both non-nullable: apply directly.-instance- {-# OVERLAPPABLE #-}- (Columnable a, Columnable b, Columnable r, a ~ BaseType a, b ~ BaseType b) =>- NullLift2Op a b r r- where- applyNull2 f = f---- | Left nullable: 'Nothing' short-circuits.-instance- {-# OVERLAPPABLE #-}- (Columnable a, Columnable b, Columnable r, Columnable (Maybe r), b ~ BaseType b) =>- NullLift2Op (Maybe a) b r (Maybe r)- where- applyNull2 _ Nothing _ = Nothing- applyNull2 f (Just x) y = Just (f x y)---- | Right nullable: 'Nothing' short-circuits.-instance- {-# OVERLAPPABLE #-}- (Columnable a, Columnable b, Columnable r, Columnable (Maybe r), a ~ BaseType a) =>- NullLift2Op a (Maybe b) r (Maybe r)- where- applyNull2 _ _ Nothing = Nothing- applyNull2 f x (Just y) = Just (f x y)---- | Both nullable: either 'Nothing' short-circuits.-instance- {-# OVERLAPPING #-}- (Columnable a, Columnable b, Columnable r, Columnable (Maybe r)) =>- NullLift2Op (Maybe a) (Maybe b) r (Maybe r)- where- applyNull2 _ Nothing _ = Nothing- applyNull2 _ _ Nothing = Nothing- applyNull2 f (Just x) (Just y) = Just (f x y)---- ------------------------------------------------------------------------------ Numeric widening--- -----------------------------------------------------------------------------{- | Widen two numeric base types to their promoted common type.--When @a ~ b@ the coercions are identity; otherwise one operand is widened-(e.g. 'Int' → 'Double').--}-class (Columnable (Promote a b)) => NumericWidenOp a b where- widen1 :: a -> Promote a b- widen2 :: b -> Promote a b---- | Same type: identity coercions.-instance {-# OVERLAPPING #-} (Columnable a) => NumericWidenOp a a where- widen1 = id- widen2 = id--instance NumericWidenOp Int Double where widen1 = fromIntegral; widen2 = id-instance NumericWidenOp Double Int where- widen1 = id- widen2 = fromIntegral-instance NumericWidenOp Float Double where widen1 = realToFrac; widen2 = id-instance NumericWidenOp Double Float where- widen1 = id- widen2 = realToFrac-instance NumericWidenOp Int32 Float where widen1 = fromIntegral; widen2 = id-instance NumericWidenOp Float Int32 where- widen1 = id- widen2 = fromIntegral-instance NumericWidenOp Int32 Double where widen1 = fromIntegral; widen2 = id-instance NumericWidenOp Double Int32 where- widen1 = id- widen2 = fromIntegral-instance NumericWidenOp Int64 Float where widen1 = fromIntegral; widen2 = id-instance NumericWidenOp Float Int64 where- widen1 = id- widen2 = fromIntegral-instance NumericWidenOp Int64 Double where widen1 = fromIntegral; widen2 = id-instance NumericWidenOp Double Int64 where- widen1 = id- widen2 = fromIntegral---- | Apply an arithmetic function after widening both operands to their common type.-widenArithOp ::- forall a b.- (NumericWidenOp a b) =>- (Promote a b -> Promote a b -> Promote a b) ->- a ->- b ->- Promote a b-widenArithOp f x y = f (widen1 @a @b x) (widen2 @a @b y)---- | Apply a comparison function after widening both operands to their common type.-widenCmpOp ::- forall a b.- (NumericWidenOp a b) =>- (Promote a b -> Promote a b -> Bool) ->- a ->- b ->- Bool-widenCmpOp f x y = f (widen1 @a @b x) (widen2 @a @b y)---- | Result type of a widening binary operator, accounting for nullable wrappers.-type WidenResult a b = NullLift2Result a b (Promote (BaseType a) (BaseType b))---- ------------------------------------------------------------------------------ Division widening (integral × integral → Double)--- -----------------------------------------------------------------------------{- | Like 'NumericWidenOp' but uses 'PromoteDiv': integral×integral → Double.-Floating types still dominate (Double > Float), and any two integral types-(same or mixed) are both widened to Double.--}-class (Columnable (PromoteDiv a b)) => DivWidenOp a b where- divWiden1 :: a -> PromoteDiv a b- divWiden2 :: b -> PromoteDiv a b---- Floating same-type (identity)-instance DivWidenOp Double Double where divWiden1 = id; divWiden2 = id-instance DivWidenOp Float Float where divWiden1 = id; divWiden2 = id---- Mixed Double/Float-instance DivWidenOp Double Float where divWiden1 = id; divWiden2 = realToFrac-instance DivWidenOp Float Double where divWiden1 = realToFrac; divWiden2 = id---- Double beats integral-instance DivWidenOp Double Int where divWiden1 = id; divWiden2 = fromIntegral-instance DivWidenOp Int Double where divWiden1 = fromIntegral; divWiden2 = id-instance DivWidenOp Double Int32 where divWiden1 = id; divWiden2 = fromIntegral-instance DivWidenOp Int32 Double where divWiden1 = fromIntegral; divWiden2 = id-instance DivWidenOp Double Int64 where divWiden1 = id; divWiden2 = fromIntegral-instance DivWidenOp Int64 Double where divWiden1 = fromIntegral; divWiden2 = id---- Float beats integral-instance DivWidenOp Float Int where divWiden1 = id; divWiden2 = fromIntegral-instance DivWidenOp Int Float where divWiden1 = fromIntegral; divWiden2 = id-instance DivWidenOp Float Int32 where divWiden1 = id; divWiden2 = fromIntegral-instance DivWidenOp Int32 Float where divWiden1 = fromIntegral; divWiden2 = id-instance DivWidenOp Float Int64 where divWiden1 = id; divWiden2 = fromIntegral-instance DivWidenOp Int64 Float where divWiden1 = fromIntegral; divWiden2 = id---- Integral × integral → Double-instance DivWidenOp Int Int where- divWiden1 = fromIntegral- divWiden2 = fromIntegral-instance DivWidenOp Int32 Int32 where- divWiden1 = fromIntegral- divWiden2 = fromIntegral-instance DivWidenOp Int64 Int64 where- divWiden1 = fromIntegral- divWiden2 = fromIntegral-instance DivWidenOp Int Int32 where- divWiden1 = fromIntegral- divWiden2 = fromIntegral-instance DivWidenOp Int32 Int where- divWiden1 = fromIntegral- divWiden2 = fromIntegral-instance DivWidenOp Int Int64 where- divWiden1 = fromIntegral- divWiden2 = fromIntegral-instance DivWidenOp Int64 Int where- divWiden1 = fromIntegral- divWiden2 = fromIntegral-instance DivWidenOp Int32 Int64 where- divWiden1 = fromIntegral- divWiden2 = fromIntegral-instance DivWidenOp Int64 Int32 where- divWiden1 = fromIntegral- divWiden2 = fromIntegral---- | Apply an arithmetic function after widening both operands via 'PromoteDiv'.-divArithOp ::- forall a b.- (DivWidenOp a b) =>- (PromoteDiv a b -> PromoteDiv a b -> PromoteDiv a b) ->- a ->- b ->- PromoteDiv a b-divArithOp f x y = f (divWiden1 @a @b x) (divWiden2 @a @b y)---- | Result type of a division-widening binary operator, accounting for nullable wrappers.-type WidenResultDiv a b =- NullLift2Result a b (PromoteDiv (BaseType a) (BaseType b))
− src-internal/DataFrame/Internal/PackedText.hs
@@ -1,238 +0,0 @@-{-# LANGUAGE BangPatterns #-}--{- | Packed-text payload + byte-slice primitives. A 'PackedTextData' shares one-UTF-8 byte buffer across all rows of a string column, with @n+1@ row offsets, so-no per-row 'Data.Text.Text' header is materialized until decode is demanded.-Offsets and selection vectors are stored 'Int32' whenever their values fit-(Arrow-style), halving the per-row footprint of large string columns.--}-module DataFrame.Internal.PackedText (- PackedTextData (..),- PackedOffsets (..),- PackedSel (..),- offAt,- offCount,- selAt,- selLength,- mkPackedContiguous,- mkPackedContiguous32,- mkOffsets,- mkSel,- packedGather,- packedTake,- packedRowOffsets,- packedLength,- packedSlice,- packedIndexText,- sliceEqBytes,- sliceCmpBytes,-) where--import qualified Data.Text as T-import qualified Data.Text.Array as A-import qualified Data.Vector.Unboxed as VU--import Data.Int (Int32)-import Data.Ord (comparing)-import Data.Text.Internal (Text (Text))-import DataFrame.Internal.Utf8 (isValidUtf8Slice, lenientDecodeSlice)--{- | Row byte-offsets, physically 'Int32' when every value fits (total buffer-bytes < 2^31) and 'Int' otherwise. Values are non-negative byte positions.--}-data PackedOffsets- = Offs32 {-# UNPACK #-} !(VU.Vector Int32)- | Offs64 {-# UNPACK #-} !(VU.Vector Int)---- | Offset at index @i@, widened to 'Int'.-offAt :: PackedOffsets -> Int -> Int-offAt (Offs32 v) i = fromIntegral (VU.unsafeIndex v i)-offAt (Offs64 v) i = VU.unsafeIndex v i-{-# INLINE offAt #-}---- | Number of offset entries (row count + 1).-offCount :: PackedOffsets -> Int-offCount (Offs32 v) = VU.length v-offCount (Offs64 v) = VU.length v-{-# INLINE offCount #-}--{- | A selection layer mapping logical rows to base rows; @-1@ marks an-invalid/null row. 'Int32' when the base row count fits.--}-data PackedSel- = Sel32 {-# UNPACK #-} !(VU.Vector Int32)- | Sel64 {-# UNPACK #-} !(VU.Vector Int)---- | Base row for logical row @i@ (may be @-1@).-selAt :: PackedSel -> Int -> Int-selAt (Sel32 v) i = fromIntegral (VU.unsafeIndex v i)-selAt (Sel64 v) i = VU.unsafeIndex v i-{-# INLINE selAt #-}--selLength :: PackedSel -> Int-selLength (Sel32 v) = VU.length v-selLength (Sel64 v) = VU.length v-{-# INLINE selLength #-}--{- | A shared UTF-8 byte buffer plus @n+1@ row offsets (base row @r@ spans bytes-@[offsets!r, offsets!(r+1))@); validity lives in the column's bitmap. @ptSel@ is-an optional selection layer letting a gather/join/sort result share the buffer.--@ptCanonicalSel@ marks a selection that is a canonical dictionary encoding:-equal byte slices always map to the same base row (codes). Set by dictionary-compaction; preserved by gather/take over an already-canonical selection (a-row keeps its code); 'False' for a gather over an unselected base, where two-logical rows can select different but equal-byted base rows. Grouping keys on-codes directly when it holds.--}-data PackedTextData = PackedTextData- { ptBytes :: {-# UNPACK #-} !A.Array- , ptOffsets :: !PackedOffsets- , ptSel :: !(Maybe PackedSel)- , ptCanonicalSel :: !Bool- }--int32Max :: Int-int32Max = fromIntegral (maxBound :: Int32)---- | Narrow an 'Int' offset vector when the final offset (total bytes) fits.-mkOffsets :: VU.Vector Int -> PackedOffsets-mkOffsets offs- | not (VU.null offs) && VU.last offs <= int32Max =- Offs32 (VU.map fromIntegral offs)- | otherwise = Offs64 offs-{-# INLINE mkOffsets #-}--{- | Narrow an 'Int' base-row vector (@-1@ sentinels allowed) when the base-row count fits in 'Int32'.--}-mkSel :: Int -> VU.Vector Int -> PackedSel-mkSel base rows- | base <= int32Max = Sel32 (VU.map fromIntegral rows)- | otherwise = Sel64 rows-{-# INLINE mkSel #-}---- | Build a contiguous packed payload (no selection): the freeze-path shape.-mkPackedContiguous :: A.Array -> VU.Vector Int -> PackedTextData-mkPackedContiguous arr offs = PackedTextData arr (mkOffsets offs) Nothing False-{-# INLINE mkPackedContiguous #-}---- | 'mkPackedContiguous' from offsets already produced at 'Int32' width.-mkPackedContiguous32 :: A.Array -> VU.Vector Int32 -> PackedTextData-mkPackedContiguous32 arr offs = PackedTextData arr (Offs32 offs) Nothing False-{-# INLINE mkPackedContiguous32 #-}--{- | Reindex a packed payload by a selection vector, sharing the byte buffer;-logical row @i@ becomes base row @indices!i@. A negative or out-of-range index-decodes to the empty slice. Composes with an existing selection; canonicality-survives composition (a kept row keeps its code) but not a first selection-over the unselected base.--}-packedGather :: VU.Vector Int -> PackedTextData -> PackedTextData-packedGather indices (PackedTextData arr offs msel canon) =- let !base = offCount offs - 1- clamp r = if r >= 0 && r < base then r else -1- (sel', canon') = case msel of- Nothing -> (VU.map clamp indices, False)- Just s ->- let !sn = selLength s- in ( VU.map- (\i -> if i >= 0 && i < sn then clamp (selAt s i) else -1)- indices- , canon- )- in PackedTextData arr offs (Just (mkSel base sel')) canon'-{-# INLINE packedGather #-}--{- | Take the first @k@ logical rows, sharing the byte buffer via a capped-selection layer. O(k), no byte copy or decode — cheap @take@/display on a-large packed column.--}-packedTake :: Int -> PackedTextData -> PackedTextData-packedTake k (PackedTextData arr offs msel canon) =- let !base = offCount offs - 1- !k' = max 0 k- (sel', canon') = case msel of- Just (Sel32 s) -> (Sel32 (VU.take k' s), canon)- Just (Sel64 s) -> (Sel64 (VU.take k' s), canon)- Nothing -> (mkSel base (VU.enumFromN 0 (min k' base)), False)- in PackedTextData arr offs (Just sel') canon'-{-# INLINE packedTake #-}---- | Map a logical row index to its base row, honoring any selection layer.-baseRow :: PackedTextData -> Int -> Int-baseRow (PackedTextData _ _ Nothing _) i = i-baseRow (PackedTextData _ _ (Just sel) _) i = selAt sel i-{-# INLINE baseRow #-}---- | Row count: @length sel@ when selected, else @length offsets - 1@.-packedLength :: PackedTextData -> Int-packedLength (PackedTextData _ offs Nothing _) = offCount offs - 1-packedLength (PackedTextData _ _ (Just sel) _) = selLength sel-{-# INLINE packedLength #-}---- | Raw byte slice for logical row @i@: @(buffer, offset, length)@. The hot accessor.-packedSlice :: PackedTextData -> Int -> (A.Array, Int, Int)-packedSlice p@(PackedTextData arr offs _ _) i =- let !r = baseRow p i- in if r < 0- then (arr, 0, 0)- else- let o = offAt offs r in (arr, o, offAt offs (r + 1) - o)-{-# INLINE packedSlice #-}--{- | The shared buffer + contiguous @n+1@ offsets when the payload is the-unselected base; a selected (gathered) payload returns 'Nothing' (its rows are-non-contiguous). Lets contiguous consumers skip the selection indirection.--}-packedRowOffsets :: PackedTextData -> Maybe (A.Array, PackedOffsets)-packedRowOffsets (PackedTextData arr offs Nothing _) = Just (arr, offs)-packedRowOffsets _ = Nothing-{-# INLINE packedRowOffsets #-}--{- | On-demand single 'Data.Text.Text' for row @i@, using the same-validate-or-lenient decode as the freeze path so output is bit-identical.--}-packedIndexText :: PackedTextData -> Int -> T.Text-packedIndexText p i =- let (arr, o, l) = packedSlice p i- in decodeField arr o l-{-# INLINE packedIndexText #-}---- Decode one field exactly as the boxed freeze path does per row.-decodeField :: A.Array -> Int -> Int -> T.Text-decodeField arr o l- | l == 0 = T.empty- | isValidUtf8Slice arr o l = Text arr o l- | otherwise = lenientDecodeSlice arr o l-{-# INLINE decodeField #-}--{- | Byte-wise equality of two slices. UTF-8 is injective on valid scalar-sequences and lenient decode is deterministic, so this agrees with-@Text@'s '==' on the decoded values.--}-sliceEqBytes :: A.Array -> Int -> Int -> A.Array -> Int -> Int -> Bool-sliceEqBytes a ao al b bo bl- | al /= bl = False- | otherwise = go 0- where- go !k- | k >= al = True- | A.unsafeIndex a (ao + k) == A.unsafeIndex b (bo + k) = go (k + 1)- | otherwise = False-{-# INLINE sliceEqBytes #-}--{- | Unsigned byte-lexicographic comparison (memcmp semantics). For-well-formed UTF-8 this matches 'Data.Text.compare' exactly, since UTF-8-byte order equals codepoint order for all valid scalars.--}-sliceCmpBytes :: A.Array -> Int -> Int -> A.Array -> Int -> Int -> Ordering-sliceCmpBytes a ao al b bo bl = go 0- where- !m = min al bl- go !k- | k >= m = compare al bl- | otherwise = case comparing id (A.unsafeIndex a (ao + k)) (A.unsafeIndex b (bo + k)) of- EQ -> go (k + 1)- r -> r-{-# INLINE sliceCmpBytes #-}
− src-internal/DataFrame/Internal/ParRadixSort.hs
@@ -1,272 +0,0 @@-{-# LANGUAGE BangPatterns #-}-{-# LANGUAGE ScopedTypeVariables #-}--{- | Parallel stable sort of row indices by ascending unsigned order of a per-row-'Int' hash, used by the join build side. A counting sort buckets rows into-key-ordered partitions that workers LSD-radix-sort in parallel, with no merge step.--}-module DataFrame.Internal.ParRadixSort (- parSortByHash,- parSortThreshold,-) where--import Control.Concurrent (forkIO, getNumCapabilities)-import Control.Concurrent.MVar (newEmptyMVar, putMVar, takeMVar)-import Control.Exception (SomeException, throwIO, try)-import Control.Monad (forM_, when)-import Data.Bits (countLeadingZeros, unsafeShiftR, (.&.))-import Data.IORef (atomicModifyIORef', newIORef)-import qualified Data.Vector.Unboxed as VU-import qualified Data.Vector.Unboxed.Mutable as VUM-import Data.Word (Word64)-import DataFrame.Internal.RadixRank (sortKey)-import System.IO.Unsafe (unsafePerformIO)--{- | Below this many rows the partition/fork overhead is not worth it; the-caller's sequential LSD radix path is used instead.--}-parSortThreshold :: Int-parSortThreshold = 500000--capabilities :: Int-capabilities = unsafePerformIO getNumCapabilities-{-# NOINLINE capabilities #-}--{- | Top-bits partition index of a hash: the high @64 - shift@ bits of its-unsigned 'sortKey'. Ascending partition order equals ascending key order.--}-partIx :: Int -> Int -> Int-partIx shift h = fromIntegral ((fromIntegral (sortKey h) :: Word64) `unsafeShiftR` shift)-{-# INLINE partIx #-}---- | Number of partitions: a power of two, at least @4 * caps@, floored at 256.-numPartitionsFor :: Int -> Int-numPartitionsFor caps = go 1- where- target = max 256 (4 * caps)- go p- | p >= target = p- | otherwise = go (p * 2)---- | @floor (log2 x)@ for a power-of-two @x@.-intLog2 :: Int -> Int-intLog2 x = 63 - countLeadingZeros x-{-# INLINE intLog2 #-}--{- | Parallel stable sort of @[0, n)@ by ascending unsigned hash order. See the-module header for the ordering contract.--}-parSortByHash :: Int -> VU.Vector Int -> (VU.Vector Int, VU.Vector Int)-parSortByHash n hashes- | n <= 1 =- (hashes, VU.enumFromN 0 n)- | n < parSortThreshold || capabilities <= 1 =- seqSortByHash n hashes- | otherwise = unsafePerformIO (parSortByHashIO n hashes)-{-# NOINLINE parSortByHash #-}------------------------------------------------------------------------------------ Sequential LSD radix sort (also the per-partition worker kernel)----------------------------------------------------------------------------------{- | Stable LSD radix sort of @[0, n)@ by ascending 'sortKey' of their hash, 8-bits per pass over the full 64-bit key. Returns @(sortedHashes, sortedIndices)@.--}-seqSortByHash :: Int -> VU.Vector Int -> (VU.Vector Int, VU.Vector Int)-seqSortByHash n hashes = unsafePerformIO $ do- keysA <- VUM.new n- orderA <- VUM.new n- let seed !i- | i >= n = pure ()- | otherwise = do- VUM.unsafeWrite keysA i (sortKey (VU.unsafeIndex hashes i))- VUM.unsafeWrite orderA i i- seed (i + 1)- seed 0- keysB <- VUM.new n- orderB <- VUM.new n- radixPasses n keysA orderA keysB orderB- order <- VU.unsafeFreeze orderA- pure (VU.unsafeBackpermute hashes order, order)--{- | Run all eight stable 8-bit LSD passes, ping-ponging between the two-key/order buffer pairs so the sorted order lands back in @(keysA, orderA)@.-@keysA[i]@ must already hold @sortKey (hash of orderA[i])@ on entry.--}-radixPasses ::- Int ->- VUM.IOVector Int ->- VUM.IOVector Int ->- VUM.IOVector Int ->- VUM.IOVector Int ->- IO ()-radixPasses n keysA orderA keysB orderB = do- counts <- VUM.new 256- let pass ::- Int ->- VUM.IOVector Int ->- VUM.IOVector Int ->- VUM.IOVector Int ->- VUM.IOVector Int ->- IO ()- pass !shiftBits !srcK !srcO !dstK !dstO = do- VUM.set counts 0- let count !i- | i >= n = pure ()- | otherwise = do- k <- VUM.unsafeRead srcK i- let !b = (k `unsafeShiftR` shiftBits) .&. 0xff- VUM.unsafeRead counts b >>= VUM.unsafeWrite counts b . (+ 1)- count (i + 1)- count 0- let scan !b !acc- | b >= 256 = pure ()- | otherwise = do- c <- VUM.unsafeRead counts b- VUM.unsafeWrite counts b acc- scan (b + 1) (acc + c)- scan 0 0- let place !i- | i >= n = pure ()- | otherwise = do- k <- VUM.unsafeRead srcK i- o <- VUM.unsafeRead srcO i- let !b = (k `unsafeShiftR` shiftBits) .&. 0xff- pos <- VUM.unsafeRead counts b- VUM.unsafeWrite counts b (pos + 1)- VUM.unsafeWrite dstK pos k- VUM.unsafeWrite dstO pos o- place (i + 1)- place 0- pass 0 keysA orderA keysB orderB- pass 8 keysB orderB keysA orderA- pass 16 keysA orderA keysB orderB- pass 24 keysB orderB keysA orderA- pass 32 keysA orderA keysB orderB- pass 40 keysB orderB keysA orderA- pass 48 keysA orderA keysB orderB- pass 56 keysB orderB keysA orderA------------------------------------------------------------------------------------ Parallel path: counting-sort partition, then per-partition sort in parallel----------------------------------------------------------------------------------parSortByHashIO :: Int -> VU.Vector Int -> IO (VU.Vector Int, VU.Vector Int)-parSortByHashIO n hashes = do- caps <- getNumCapabilities- let !p = numPartitionsFor caps- !shift = 64 - intLog2 p- (partStart, partRows) <- partitionRows n hashes p shift- outOrder <- VUM.new n- outKeys <- VUM.new n- sortPartitions caps p partStart partRows hashes outOrder outKeys- order <- VU.unsafeFreeze outOrder- pure (VU.unsafeBackpermute hashes order, order)--{- | Bucket every row index into its top-bits partition by a counting sort.-Returns the exclusive prefix sum @partStart@ (length @p+1@, @partStart[p] == n@)-and the row indices laid out partition-by-partition in ascending key order.--}-partitionRows ::- Int -> VU.Vector Int -> Int -> Int -> IO (VU.Vector Int, VU.Vector Int)-partitionRows n hashes p shift = do- counts <- VUM.replicate (p + 1) (0 :: Int)- let countLoop !i- | i >= n = pure ()- | otherwise = do- let !pp = partIx shift (VU.unsafeIndex hashes i)- c <- VUM.unsafeRead counts pp- VUM.unsafeWrite counts pp (c + 1)- countLoop (i + 1)- countLoop 0- partStartM <- VUM.new (p + 1)- let scan !k !acc- | k > p = pure ()- | otherwise = do- VUM.unsafeWrite partStartM k acc- c <- if k < p then VUM.unsafeRead counts k else pure 0- scan (k + 1) (acc + c)- scan 0 0- cursor <- VUM.new p- forM_ [0 .. p - 1] $ \k -> VUM.unsafeRead partStartM k >>= VUM.unsafeWrite cursor k- rowsM <- VUM.new (max 1 n)- let place !i- | i >= n = pure ()- | otherwise = do- let !pp = partIx shift (VU.unsafeIndex hashes i)- pos <- VUM.unsafeRead cursor pp- VUM.unsafeWrite rowsM pos i- VUM.unsafeWrite cursor pp (pos + 1)- place (i + 1)- place 0- partStart <- VU.unsafeFreeze partStartM- partRows <- VU.unsafeFreeze rowsM- pure (partStart, partRows)--{- | Stable-sort each partition by full key, writing sorted original indices-into @outOrder@ and their hashes into @outKeys@ at the partition's slot range.-Forks @caps@ workers that pull partition indices off a shared atomic counter.-Within a partition the counting sort already left rows in ascending original-order, so the LSD radix sort's stability reproduces the global @(key, row)@-order. Partitions below two elements are already sorted (counting sort kept-original order) and are copied directly.--}-sortPartitions ::- Int ->- Int ->- VU.Vector Int ->- VU.Vector Int ->- VU.Vector Int ->- VUM.IOVector Int ->- VUM.IOVector Int ->- IO ()-sortPartitions caps p partStart partRows hashes outOrder outKeys = do- next <- newIORef 0- let sortOne !pp = do- let !s = VU.unsafeIndex partStart pp- !e = VU.unsafeIndex partStart (pp + 1)- !sz = e - s- when (sz > 0) $- if sz == 1- then do- let !r = VU.unsafeIndex partRows s- VUM.unsafeWrite outOrder s r- VUM.unsafeWrite outKeys s (VU.unsafeIndex hashes r)- else do- keysA <- VUM.new sz- orderA <- VUM.new sz- let seed !i- | i >= sz = pure ()- | otherwise = do- let !r = VU.unsafeIndex partRows (s + i)- VUM.unsafeWrite keysA i (sortKey (VU.unsafeIndex hashes r))- VUM.unsafeWrite orderA i r- seed (i + 1)- seed 0- keysB <- VUM.new sz- orderB <- VUM.new sz- radixPasses sz keysA orderA keysB orderB- let emit !i- | i >= sz = pure ()- | otherwise = do- o <- VUM.unsafeRead orderA i- VUM.unsafeWrite outOrder (s + i) o- VUM.unsafeWrite outKeys (s + i) (VU.unsafeIndex hashes o)- emit (i + 1)- emit 0- worker = do- i <- atomicModifyIORef' next (\j -> (j + 1, j))- when (i < p) $ sortOne i >> worker- forkJoin_ (replicate caps worker)---- | Run each action on its own thread; rethrow the first failure (in order).-forkJoin_ :: [IO ()] -> IO ()-forkJoin_ actions = do- vars <- mapM spawn actions- results <- mapM takeMVar vars- mapM_ (either (throwIO :: SomeException -> IO ()) pure) results- where- spawn act = do- var <- newEmptyMVar- _ <- forkIO (try act >>= putMVar var)- pure var
− src-internal/DataFrame/Internal/Pretty.hs
@@ -1,129 +0,0 @@-{- | A minimal Wadler/Leijen-style document combinator and width-aware renderer.-A 'Doc' describes a layout abstractly; 'render' chooses where soft breaks become-newlines to fit a target width. 'Group' lays a region flat when it fits.--}-module DataFrame.Internal.Pretty (- Doc,- text,- line,- hardline,- nest,- group,- (<+>),- hcat,- punctuate,- parens,- parensWhenBroken,- defaultWidth,- render,-) where--data Doc- = Empty- | Text String- | Line- | Cat Doc Doc- | Nest Int Doc- | Group Doc- | Hard- | Alt Doc Doc--instance Semigroup Doc where- (<>) = Cat--instance Monoid Doc where- mempty = Empty---- | A literal chunk of text. Must not contain newlines (use 'line'/'hardline').-text :: String -> Doc-text = Text--{- | A soft break: a single space when its enclosing 'group' fits the width,-otherwise a newline + current indentation.--}-line :: Doc-line = Line---- | A hard break that never flattens; any enclosing 'group' is forced to break.-hardline :: Doc-hardline = Hard---- | Add @k@ spaces to the indentation applied at line breaks inside @d@.-nest :: Int -> Doc -> Doc-nest = Nest---- | Lay the document out flat if it fits the remaining width, broken otherwise.-group :: Doc -> Doc-group = Group---- | Concatenate two documents separated by a single space.-(<+>) :: Doc -> Doc -> Doc-x <+> y = x <> Text " " <> y--infixr 6 <+>--hcat :: [Doc] -> Doc-hcat = mconcat---- | Append @sep@ after every element but the last.-punctuate :: Doc -> [Doc] -> [Doc]-punctuate _ [] = []-punctuate _ [d] = [d]-punctuate sep (d : ds) = (d <> sep) : punctuate sep ds--parens :: Doc -> Doc-parens d = Text "(" <> d <> Text ")"--{- | Render @d@ bare when it fits flat on the current line, wrapped in parens when-it must break across lines. Keeps operator grouping unambiguous once a-sub-expression wraps, without parenthesis noise on one-line expressions.--}-parensWhenBroken :: Doc -> Doc-parensWhenBroken d = Group (Alt d (parens d))--defaultWidth :: Int-defaultWidth = 80--data Mode = Flat | Break---- | Render a document, breaking soft lines so output fits @width@ columns.-render :: Int -> Doc -> String-render width doc = layout 0 [(0, Break, doc)]- where- layout :: Int -> [(Int, Mode, Doc)] -> String- layout _ [] = ""- layout col ((i, m, d) : rest) = case d of- Empty -> layout col rest- Text s -> s ++ layout (col + length s) rest- Cat x y -> layout col ((i, m, x) : (i, m, y) : rest)- Nest j x -> layout col ((i + j, m, x) : rest)- Line -> case m of- Flat -> ' ' : layout (col + 1) rest- Break -> '\n' : replicate i ' ' ++ layout i rest- Hard -> '\n' : replicate i ' ' ++ layout i rest- Group x ->- if fits (width - col) ((i, Flat, x) : rest)- then layout col ((i, Flat, x) : rest)- else layout col ((i, Break, x) : rest)- Alt flat broken -> case m of- Flat -> layout col ((i, Flat, flat) : rest)- Break -> layout col ((i, Break, broken) : rest)-- fits :: Int -> [(Int, Mode, Doc)] -> Bool- fits w _ | w < 0 = False- fits _ [] = True- fits w ((i, m, d) : rest) = case d of- Empty -> fits w rest- Text s -> fits (w - length s) rest- Cat x y -> fits w ((i, m, x) : (i, m, y) : rest)- Nest j x -> fits w ((i + j, m, x) : rest)- Line -> case m of- Flat -> fits (w - 1) rest- Break -> True- Hard -> case m of- Flat -> False- Break -> True- Group x -> fits w ((i, Flat, x) : rest)- Alt flat broken -> case m of- Flat -> fits w ((i, Flat, flat) : rest)- Break -> fits w ((i, Break, broken) : rest)
− src-internal/DataFrame/Internal/RadixRank.hs
@@ -1,101 +0,0 @@-{-# LANGUAGE BangPatterns #-}-{-# LANGUAGE ScopedTypeVariables #-}--{- | Stable rank of a set of group representatives by ascending unsigned hash-order. Shared by the sequential and parallel group-by canonical-ordering steps-so they stay bit-for-bit identical. @O(ng)@ stable LSD radix sort.--}-module DataFrame.Internal.RadixRank (- rankByHash,- sortKey,-) where--import Control.Monad (when)-import Control.Monad.Primitive (PrimMonad)-import Data.Bits (unsafeShiftR, (.&.))-import qualified Data.Vector.Unboxed as VU-import qualified Data.Vector.Unboxed.Mutable as VUM-import Data.Word (Word64)--{- | Unsigned sort key of a hash: ascending 'Word64' order of @sortKey h@ equals-ascending signed-'Int' order of @h@. Reinterpreted to 'Int' for the byte-wise-radix passes (the byte mask makes the sign extension irrelevant).--}-sortKey :: Int -> Int-sortKey h = fromIntegral (fromIntegral h + 0x8000000000000000 :: Word64)-{-# INLINE sortKey #-}---- | See the module header. @readHash@ supplies the hash of local group @gid@.-rankByHash ::- forall m. (PrimMonad m) => (Int -> m Int) -> Int -> m (VU.Vector Int)-rankByHash readHash ng = do- rankM <- VUM.new (max 1 ng)- if ng <= 1- then when (ng == 1) (VUM.unsafeWrite rankM 0 0)- else do- keysA <- VUM.new ng- orderA <- VUM.new ng- let seed !i- | i >= ng = pure ()- | otherwise = do- h <- readHash i- VUM.unsafeWrite keysA i (sortKey h)- VUM.unsafeWrite orderA i i- seed (i + 1)- seed 0- keysB <- VUM.new ng- orderB <- VUM.new ng- counts <- VUM.new 256- let pass ::- Int ->- VUM.MVector (VUM.PrimState m) Int ->- VUM.MVector (VUM.PrimState m) Int ->- VUM.MVector (VUM.PrimState m) Int ->- VUM.MVector (VUM.PrimState m) Int ->- m ()- pass !shiftBits !srcK !srcO !dstK !dstO = do- VUM.set counts 0- let count !i- | i >= ng = pure ()- | otherwise = do- k <- VUM.unsafeRead srcK i- let !b = (k `unsafeShiftR` shiftBits) .&. 0xff- VUM.unsafeRead counts b >>= VUM.unsafeWrite counts b . (+ 1)- count (i + 1)- count 0- let scan !b !acc- | b >= 256 = pure ()- | otherwise = do- c <- VUM.unsafeRead counts b- VUM.unsafeWrite counts b acc- scan (b + 1) (acc + c)- scan 0 0- let place !i- | i >= ng = pure ()- | otherwise = do- k <- VUM.unsafeRead srcK i- o <- VUM.unsafeRead srcO i- let !b = (k `unsafeShiftR` shiftBits) .&. 0xff- pos <- VUM.unsafeRead counts b- VUM.unsafeWrite counts b (pos + 1)- VUM.unsafeWrite dstK pos k- VUM.unsafeWrite dstO pos o- place (i + 1)- place 0- pass 0 keysA orderA keysB orderB- pass 8 keysB orderB keysA orderA- pass 16 keysA orderA keysB orderB- pass 24 keysB orderB keysA orderA- pass 32 keysA orderA keysB orderB- pass 40 keysB orderB keysA orderA- pass 48 keysA orderA keysB orderB- pass 56 keysB orderB keysA orderA- let inv !r- | r >= ng = pure ()- | otherwise = do- g <- VUM.unsafeRead orderA r- VUM.unsafeWrite rankM g r- inv (r + 1)- inv 0- VU.unsafeFreeze rankM-{-# INLINEABLE rankByHash #-}
src-internal/DataFrame/Internal/Row.hs view
@@ -21,10 +21,22 @@ import Data.Type.Equality (TestEquality (..)) import Data.Typeable (Typeable, type (:~:) (..)) import DataFrame.Errors (DataFrameException (..), TypeErrorContext (..))-import DataFrame.Internal.Column-import DataFrame.Internal.DataFrame+import DataFrame.Internal.Column (+ Column (..),+ Columnable,+ columnLength,+ fromList,+ fromVector,+ materializeMerged,+ sliceColumn,+ )+import DataFrame.Internal.Column.Bitmap (Bitmap, bitmapTestBit)+import DataFrame.Internal.Data.PackedText (packedIndexText, packedLength)+import DataFrame.Internal.DataFrame (+ DataFrame (columnIndices, dataframeDimensions),+ getColumn,+ ) import DataFrame.Internal.Expression (Expr (..))-import DataFrame.Internal.PackedText (packedIndexText, packedLength) import Type.Reflection (TypeRep, typeOf, typeRep) data Any where@@ -91,9 +103,9 @@ Nothing -> throw (mismatchAt r (typeRep @b) (typeRep @a)) maybes = zipWith collect [0 :: Int ..] cells in if any isNothing maybes- then fromMaybeVec (V.fromList maybes)+ then fromVector (V.fromList maybes) else fromList (catMaybes maybes)- _ -> fromMaybeVec (V.fromList (map (const (Nothing :: Maybe T.Text)) cells))+ _ -> fromVector (V.fromList (map (const (Nothing :: Maybe T.Text)) cells)) where cells = zipWith cellAt [0 :: Int ..] rows cellAt r row = fromMaybe (throw (missingCellAt r)) (row !? i)
+ src-internal/DataFrame/Internal/Row/RowHash.hs view
@@ -0,0 +1,240 @@+{-# LANGUAGE BangPatterns #-}+{-# LANGUAGE ExplicitNamespaces #-}+{-# LANGUAGE GADTs #-}+{-# LANGUAGE LambdaCase #-}+{-# LANGUAGE ScopedTypeVariables #-}+{-# LANGUAGE TypeApplications #-}++{- | Row-hash kernels with a parallel driver, feeding grouping and the join+build/probe. Each row's hash depends only on its own bytes, so hashing disjoint+ranges in parallel is race-free and bit-identical to the sequential pass.+-}+module DataFrame.Internal.Row.RowHash (+ computeRowHashesIO,+ computeRowHashesWithIO,+ hashRowRange,+ parRowHashThreshold,+) where++import qualified Data.Text as T+import Data.Type.Equality (TestEquality (..), type (:~:) (Refl))+import qualified Data.Vector as V+import qualified Data.Vector.Unboxed as VU+import qualified Data.Vector.Unboxed.Mutable as VUM+import Type.Reflection (typeRep)++import DataFrame.Internal.Algorithms.Hash (+ fnvOffset,+ mixBytes,+ mixDouble,+ mixInt,+ mixShow,+ mixText,+ nullSalt,+ )+import DataFrame.Internal.Column (+ Column (..),+ materializeMerged,+ )+import DataFrame.Internal.Column.Bitmap (+ Bitmap,+ bitmapTestBit,+ )+import DataFrame.Internal.Column.Types (+ SBool (..),+ sFloating,+ sIntegral,+ )+import DataFrame.Internal.Control.Concurrent (parallelChunks_)+import DataFrame.Internal.Data.PackedText (+ PackedSel,+ PackedTextData (..),+ offAt,+ packedSlice,+ selAt,+ )++{- | At least this many rows make the fork/coordination overhead of the parallel+hash worth it. Below it the sequential single range is used. Matches the+grouping/join parallel thresholds so the whole pipeline switches together.+-}+parRowHashThreshold :: Int+parRowHashThreshold = 200000++{- | Compute the per-row key hash over the selected key columns of an @n@-row+frame. Forks one worker per capability over disjoint row ranges when the row+count justifies it, else hashes the single full range; output is capability-independent.++Dictionary-code hashing is disabled: this is the join entry point, and joins+hash each side by its own representation, so a canonical dict column on one+side of a join against a plain 'T.Text' (or non-canonical packed) column on+the other must byte-hash to keep both sides bucketing identically.+-}+computeRowHashesIO :: Int -> [Column] -> IO (VU.Vector Int)+computeRowHashesIO = computeRowHashesWithIO False++{- | 'computeRowHashesIO' with an explicit dictionary-code switch. When+@useDictCodes@ is 'True', a canonical dict-encoded 'PackedText' column mixes+its 'Int' code per row instead of its byte slice (equal strings share a code,+so bucketing within one frame is preserved). Only sound when every consumer of+the hashes uses the same rule — the grouping path passes 'True', joins 'False'.+-}+computeRowHashesWithIO :: Bool -> Int -> [Column] -> IO (VU.Vector Int)+computeRowHashesWithIO useDictCodes n selected = do+ mv <- VUM.unsafeNew (max 1 n)+ let runRange lo hi = hashRowRange useDictCodes mv lo hi selected+ parallelChunks_ parRowHashThreshold n runRange+ VU.unsafeFreeze (VUM.slice 0 n mv)++{- | Mix every selected column over the row range @[lo, hi)@ into @mv@, seeding+each slot with 'fnvOffset'. Must match the sequential grouping hash byte-for-byte+(at the same @useDictCodes@ setting) so grouping and joins bucket identically.+-}+hashRowRange :: Bool -> VUM.IOVector Int -> Int -> Int -> [Column] -> IO ()+hashRowRange useDictCodes mv lo hi cols = do+ seedRange mv lo hi+ mapM_ (mixColumnRange useDictCodes mv lo hi) cols++seedRange :: VUM.IOVector Int -> Int -> Int -> IO ()+seedRange mv lo hi = go lo+ where+ go !i+ | i >= hi = pure ()+ | otherwise = VUM.unsafeWrite mv i fnvOffset >> go (i + 1)++{- | Fold one column's values over @[lo, hi)@ into the running hashes. The branch+structure mirrors the sequential grouping hash: typed unboxed fast paths, then a+'mixShow' fallback, with the null bitmap mixing 'nullSalt'.+-}+mixColumnRange :: Bool -> VUM.IOVector Int -> Int -> Int -> Column -> IO ()+mixColumnRange useDictCodes mv lo hi = \case+ c@(MergedColumn _ _) -> mixColumnRange useDictCodes mv lo hi (materializeMerged c)+ UnboxedColumn ubm (v :: VU.Vector a) ->+ case testEquality (typeRep @a) (typeRep @Int) of+ Just Refl -> unboxedRange mv lo hi ubm mixInt v+ Nothing ->+ case testEquality (typeRep @a) (typeRep @Double) of+ Just Refl -> unboxedRange mv lo hi ubm mixDouble v+ Nothing ->+ case sIntegral @a of+ STrue ->+ unboxedRange mv lo hi ubm (\h d -> mixInt h (fromIntegral @a @Int d)) v+ SFalse ->+ case sFloating @a of+ STrue ->+ unboxedRange mv lo hi ubm (\h d -> mixDouble h (realToFrac d :: Double)) v+ SFalse ->+ unboxedRange mv lo hi ubm mixShow v+ BoxedColumn bm (v :: V.Vector a) ->+ case testEquality (typeRep @a) (typeRep @T.Text) of+ Just Refl -> boxedRange mv lo hi bm mixText v+ Nothing -> boxedRange mv lo hi bm mixShow v+ PackedText bm p -> packedRange useDictCodes mv lo hi bm p++{- | Mix an unboxed column's range, mixing 'nullSalt' at null slots. @INLINE@d to+specialise on the element type and mixing function per call site.+-}+unboxedRange ::+ (VU.Unbox a) =>+ VUM.IOVector Int ->+ Int ->+ Int ->+ Maybe Bitmap ->+ (Int -> a -> Int) ->+ VU.Vector a ->+ IO ()+unboxedRange mv lo hi ubm mix v = go lo+ where+ go !i+ | i >= hi = pure ()+ | otherwise = do+ h <- VUM.unsafeRead mv i+ let !h' = case ubm of+ Just bm | not (bitmapTestBit bm i) -> mixInt h nullSalt+ _ -> mix h (VU.unsafeIndex v i)+ VUM.unsafeWrite mv i h'+ go (i + 1)+{-# INLINE unboxedRange #-}++boxedRange ::+ VUM.IOVector Int ->+ Int ->+ Int ->+ Maybe Bitmap ->+ (Int -> a -> Int) ->+ V.Vector a ->+ IO ()+boxedRange mv lo hi bm mix v = go lo+ where+ go !i+ | i >= hi = pure ()+ | otherwise = do+ h <- VUM.unsafeRead mv i+ let !h' = case bm of+ Just bm' | not (bitmapTestBit bm' i) -> mixInt h nullSalt+ _ -> mix h (V.unsafeIndex v i)+ VUM.unsafeWrite mv i h'+ go (i + 1)+{-# INLINE boxedRange #-}++{- | Mix a packed-text column's range over its raw UTF-8 byte slices. The+unselected payload is the hot path (indexes the offset vector directly); a+selected payload (a gather/join result) falls back to 'packedSlice'. When+@useDictCodes@ holds and the selection is a canonical dictionary encoding+(equal strings share a code), each row mixes its 'Int' code with one 'mixInt'+instead of walking the string bytes.+-}+packedRange ::+ Bool ->+ VUM.IOVector Int ->+ Int ->+ Int ->+ Maybe Bitmap ->+ PackedTextData ->+ IO ()+packedRange useDictCodes mv lo hi bm p =+ case ptSel p of+ Just sel | useDictCodes && ptCanonicalSel p -> codes sel+ Nothing -> contiguous (ptBytes p) (ptOffsets p)+ Just _ -> selected+ where+ valid i = case bm of+ Just bm' -> bitmapTestBit bm' i+ Nothing -> True+ contiguous !arr !offs = go lo+ where+ go !i+ | i >= hi = pure ()+ | otherwise = do+ h <- VUM.unsafeRead mv i+ let !o = offAt offs i+ !l = offAt offs (i + 1) - o+ !h' = if valid i then mixBytes h arr o l else mixInt h nullSalt+ VUM.unsafeWrite mv i h'+ go (i + 1)+ selected = go lo+ where+ go !i+ | i >= hi = pure ()+ | otherwise = do+ h <- VUM.unsafeRead mv i+ let !h' =+ if valid i+ then let (arr, o, l) = packedSlice p i in mixBytes h arr o l+ else mixInt h nullSalt+ VUM.unsafeWrite mv i h'+ go (i + 1)+ codes :: PackedSel -> IO ()+ codes !sel = go lo+ where+ go !i+ | i >= hi = pure ()+ | otherwise = do+ h <- VUM.unsafeRead mv i+ let !h' =+ if valid i+ then mixInt h (selAt sel i)+ else mixInt h nullSalt+ VUM.unsafeWrite mv i h'+ go (i + 1)+{-# INLINE packedRange #-}
− src-internal/DataFrame/Internal/RowHash.hs
@@ -1,223 +0,0 @@-{-# LANGUAGE BangPatterns #-}-{-# LANGUAGE ExplicitNamespaces #-}-{-# LANGUAGE GADTs #-}-{-# LANGUAGE LambdaCase #-}-{-# LANGUAGE ScopedTypeVariables #-}-{-# LANGUAGE TypeApplications #-}--{- | Row-hash kernels with a parallel driver, feeding grouping and the join-build/probe. Each row's hash depends only on its own bytes, so hashing disjoint-ranges in parallel is race-free and bit-identical to the sequential pass.--}-module DataFrame.Internal.RowHash (- computeRowHashesIO,- hashRowRange,- parRowHashThreshold,-) where--import Control.Concurrent (forkIO, getNumCapabilities)-import Control.Concurrent.MVar (newEmptyMVar, putMVar, takeMVar)-import Control.Exception (SomeException, throwIO, try)-import qualified Data.Text as T-import Data.Type.Equality (TestEquality (..), type (:~:) (Refl))-import qualified Data.Vector as V-import qualified Data.Vector.Unboxed as VU-import qualified Data.Vector.Unboxed.Mutable as VUM-import System.IO.Unsafe (unsafePerformIO)-import Type.Reflection (typeRep)--import DataFrame.Internal.Column (- Bitmap,- Column (..),- bitmapTestBit,- materializeMerged,- )-import DataFrame.Internal.Hash (- fnvOffset,- mixBytes,- mixDouble,- mixInt,- mixShow,- mixText,- nullSalt,- )-import DataFrame.Internal.PackedText (- PackedTextData (..),- offAt,- packedSlice,- )-import DataFrame.Internal.Types (- SBool (..),- sFloating,- sIntegral,- )--{- | At least this many rows make the fork/coordination overhead of the parallel-hash worth it. Below it the sequential single range is used. Matches the-grouping/join parallel thresholds so the whole pipeline switches together.--}-parRowHashThreshold :: Int-parRowHashThreshold = 200000--capabilities :: Int-capabilities = unsafePerformIO getNumCapabilities-{-# NOINLINE capabilities #-}--{- | Compute the per-row key hash over the selected key columns of an @n@-row-frame. Forks one worker per capability over disjoint row ranges when the row-count justifies it, else hashes the single full range; output is capability-independent.--}-computeRowHashesIO :: Int -> [Column] -> IO (VU.Vector Int)-computeRowHashesIO n selected = do- mv <- VUM.unsafeNew (max 1 n)- let runRange lo hi = hashRowRange mv lo hi selected- if n >= parRowHashThreshold && capabilities > 1- then do- let !caps = capabilities- !per = (n + caps - 1) `div` caps- spawn w = do- var <- newEmptyMVar- let !lo = min n (w * per)- !hi = min n (lo + per)- _ <- forkIO (try (runRange lo hi) >>= putMVar var)- pure var- vars <- mapM spawn [0 .. caps - 1]- rs <- mapM takeMVar vars- mapM_ (either (throwIO @SomeException) pure) rs- else runRange 0 n- VU.unsafeFreeze (VUM.slice 0 n mv)--{- | Mix every selected column over the row range @[lo, hi)@ into @mv@, seeding-each slot with 'fnvOffset'. Must match the sequential grouping hash byte-for-byte-so grouping and joins bucket identically.--}-hashRowRange :: VUM.IOVector Int -> Int -> Int -> [Column] -> IO ()-hashRowRange mv lo hi cols = do- seedRange mv lo hi- mapM_ (mixColumnRange mv lo hi) cols--seedRange :: VUM.IOVector Int -> Int -> Int -> IO ()-seedRange mv lo hi = go lo- where- go !i- | i >= hi = pure ()- | otherwise = VUM.unsafeWrite mv i fnvOffset >> go (i + 1)--{- | Fold one column's values over @[lo, hi)@ into the running hashes. The branch-structure mirrors the sequential grouping hash: typed unboxed fast paths, then a-'mixShow' fallback, with the null bitmap mixing 'nullSalt'.--}-mixColumnRange :: VUM.IOVector Int -> Int -> Int -> Column -> IO ()-mixColumnRange mv lo hi = \case- c@(MergedColumn _ _) -> mixColumnRange mv lo hi (materializeMerged c)- UnboxedColumn ubm (v :: VU.Vector a) ->- case testEquality (typeRep @a) (typeRep @Int) of- Just Refl -> unboxedRange mv lo hi ubm mixInt v- Nothing ->- case testEquality (typeRep @a) (typeRep @Double) of- Just Refl -> unboxedRange mv lo hi ubm mixDouble v- Nothing ->- case sIntegral @a of- STrue ->- unboxedRange mv lo hi ubm (\h d -> mixInt h (fromIntegral @a @Int d)) v- SFalse ->- case sFloating @a of- STrue ->- unboxedRange mv lo hi ubm (\h d -> mixDouble h (realToFrac d :: Double)) v- SFalse ->- unboxedRange mv lo hi ubm mixShow v- BoxedColumn bm (v :: V.Vector a) ->- case testEquality (typeRep @a) (typeRep @T.Text) of- Just Refl -> boxedRange mv lo hi bm mixText v- Nothing -> boxedRange mv lo hi bm mixShow v- PackedText bm p -> packedRange mv lo hi bm p--{- | Mix an unboxed column's range, mixing 'nullSalt' at null slots. @INLINE@d to-specialise on the element type and mixing function per call site.--}-unboxedRange ::- (VU.Unbox a) =>- VUM.IOVector Int ->- Int ->- Int ->- Maybe Bitmap ->- (Int -> a -> Int) ->- VU.Vector a ->- IO ()-unboxedRange mv lo hi ubm mix v = go lo- where- go !i- | i >= hi = pure ()- | otherwise = do- h <- VUM.unsafeRead mv i- let !h' = case ubm of- Just bm | not (bitmapTestBit bm i) -> mixInt h nullSalt- _ -> mix h (VU.unsafeIndex v i)- VUM.unsafeWrite mv i h'- go (i + 1)-{-# INLINE unboxedRange #-}--boxedRange ::- VUM.IOVector Int ->- Int ->- Int ->- Maybe Bitmap ->- (Int -> a -> Int) ->- V.Vector a ->- IO ()-boxedRange mv lo hi bm mix v = go lo- where- go !i- | i >= hi = pure ()- | otherwise = do- h <- VUM.unsafeRead mv i- let !h' = case bm of- Just bm' | not (bitmapTestBit bm' i) -> mixInt h nullSalt- _ -> mix h (V.unsafeIndex v i)- VUM.unsafeWrite mv i h'- go (i + 1)-{-# INLINE boxedRange #-}--{- | Mix a packed-text column's range over its raw UTF-8 byte slices. The-unselected payload is the hot path (indexes the offset vector directly); a-selected payload (a gather/join result) falls back to 'packedSlice'.--}-packedRange ::- VUM.IOVector Int ->- Int ->- Int ->- Maybe Bitmap ->- PackedTextData ->- IO ()-packedRange mv lo hi bm p =- case ptSel p of- Nothing -> contiguous (ptBytes p) (ptOffsets p)- Just _ -> selected- where- valid i = case bm of- Just bm' -> bitmapTestBit bm' i- Nothing -> True- contiguous !arr !offs = go lo- where- go !i- | i >= hi = pure ()- | otherwise = do- h <- VUM.unsafeRead mv i- let !o = offAt offs i- !l = offAt offs (i + 1) - o- !h' = if valid i then mixBytes h arr o l else mixInt h nullSalt- VUM.unsafeWrite mv i h'- go (i + 1)- selected = go lo- where- go !i- | i >= hi = pure ()- | otherwise = do- h <- VUM.unsafeRead mv i- let !h' =- if valid i- then let (arr, o, l) = packedSlice p i in mixBytes h arr o l- else mixInt h nullSalt- VUM.unsafeWrite mv i h'- go (i + 1)-{-# INLINE packedRange #-}
− src-internal/DataFrame/Internal/Simplify.hs
@@ -1,417 +0,0 @@-{-# LANGUAGE AllowAmbiguousTypes #-}-{-# LANGUAGE FlexibleContexts #-}-{-# LANGUAGE GADTs #-}-{-# LANGUAGE MultiWayIf #-}-{-# LANGUAGE OverloadedStrings #-}-{-# LANGUAGE PatternSynonyms #-}-{-# LANGUAGE ScopedTypeVariables #-}-{-# LANGUAGE TypeApplications #-}--module DataFrame.Internal.Simplify (- simplify,- simplifyPredicatePair,-- -- * Path-condition entailment (for fitted-tree pruning)- PredFact,- factTrue,- factFalse,- entails,-) where--import Control.Monad (guard)-import Data.Maybe (fromMaybe)-import Data.Type.Equality (testEquality, (:~:) (Refl))-import Type.Reflection (eqTypeRep, typeRep, (:~~:) (HRefl), pattern App)--import DataFrame.Internal.Column (Columnable)-import DataFrame.Internal.Expression (- BinaryOp,- Expr (..),- UnaryOp (unaryName),- eqExpr,- normalize,- )-import DataFrame.Operators (- NullAnd,- NullEq,- NullGeq,- NullGt,- NullLeq,- NullLt,- NullNeq,- NullOr,- (.==.),- )--simplify :: forall a. (Columnable a) => Expr a -> Expr a-simplify e- | isBoolish @a = fixpoint (10 :: Int) e- | otherwise = e- where- fixpoint 0 x = x- fixpoint n x = let x' = simplifyB x in if eqExpr x x' then x else fixpoint (n - 1) x'--isBoolish :: forall a. (Columnable a) => Bool-isBoolish =- case ( testEquality (typeRep @a) (typeRep @Bool)- , testEquality (typeRep @a) (typeRep @(Maybe Bool))- ) of- (Just Refl, _) -> True- (_, Just Refl) -> True- _ -> False--data Conn = ConnAnd | ConnOr--connOf :: forall op c b r. (BinaryOp op) => op c b r -> Maybe Conn-connOf _- | Just HRefl <- eqTypeRep (typeRep @op) (typeRep @NullAnd) = Just ConnAnd- | Just HRefl <- eqTypeRep (typeRep @op) (typeRep @NullOr) = Just ConnOr- | otherwise = Nothing--simplifyB :: forall a. (Columnable a) => Expr a -> Expr a-simplifyB expr = case expr of- Binary (op :: op c b a) l r- | Just conn <- connOf op- , Just Refl <- testEquality (typeRep @c) (typeRep @a)- , Just Refl <- testEquality (typeRep @b) (typeRep @a) ->- let l' = simplifyB l; r' = simplifyB r- in fromMaybe (Binary op l' r') (combine conn l' r')- | otherwise -> expr- Unary (op :: op b a) inner- | Just Refl <- testEquality (typeRep @a) (typeRep @Bool)- , Just Refl <- testEquality (typeRep @b) (typeRep @Bool)- , unaryName op == "not" ->- simplifyNot op (simplifyB inner)- | otherwise -> expr- If c t f ->- let c' = simplify c- t' = simplifyB t- f' = simplifyB f- in case asBoolLit c' of- Just True -> t'- Just False -> f'- Nothing- | eqExpr t' f' -> t'- | Just Refl <- testEquality (typeRep @a) (typeRep @Bool)- , asBoolLit t' == Just True- , asBoolLit f' == Just False ->- c'- | otherwise -> If c' t' f'- _ -> expr--simplifyNot :: (UnaryOp op) => op Bool Bool -> Expr Bool -> Expr Bool-simplifyNot op inner = case asBoolLit inner of- Just b -> Lit (not b)- Nothing -> case inner of- Unary (op2 :: op2 b2 Bool) inner2- | unaryName op2 == "not"- , Just Refl <- testEquality (typeRep @b2) (typeRep @Bool) ->- inner2- _ -> Unary op inner--combine :: (Columnable a) => Conn -> Expr a -> Expr a -> Maybe (Expr a)-combine ConnAnd = combineAnd-combine ConnOr = combineOr--asBoolLit :: forall a. (Columnable a) => Expr a -> Maybe Bool-asBoolLit (Lit v) =- case testEquality (typeRep @a) (typeRep @Bool) of- Just Refl -> Just v- Nothing -> case testEquality (typeRep @a) (typeRep @(Maybe Bool)) of- Just Refl -> v- Nothing -> Nothing-asBoolLit _ = Nothing--{- | Polymorphic boolean literal: @Lit b@ for @Expr Bool@, @Lit (Just b)@ for-@Expr (Maybe Bool)@.--}-litBoolish :: forall a. (Columnable a) => Bool -> Maybe (Expr a)-litBoolish v =- case testEquality (typeRep @a) (typeRep @Bool) of- Just Refl -> Just (Lit v)- Nothing -> case testEquality (typeRep @a) (typeRep @(Maybe Bool)) of- Just Refl -> Just (Lit (Just v))- Nothing -> Nothing--combineAnd :: (Columnable a) => Expr a -> Expr a -> Maybe (Expr a)-combineAnd l r- | eqExpr l r = Just l- | asBoolLit l == Just False = litBoolish False- | asBoolLit r == Just False = litBoolish False- | asBoolLit l == Just True = Just r- | asBoolLit r == Just True = Just l- | absorbs ConnOr l r = Just l- | absorbs ConnOr r l = Just r- | otherwise = simplifyPredicatePair True l r--combineOr :: (Columnable a) => Expr a -> Expr a -> Maybe (Expr a)-combineOr l r- | eqExpr l r = Just l- | asBoolLit l == Just True = litBoolish True- | asBoolLit r == Just True = litBoolish True- | asBoolLit l == Just False = Just r- | asBoolLit r == Just False = Just l- | absorbs ConnAnd l r = Just l- | absorbs ConnAnd r l = Just r- | otherwise = simplifyPredicatePair False l r--absorbs :: (Columnable a) => Conn -> Expr a -> Expr a -> Bool-absorbs conn x (Binary (op :: op c b a) ya yb)- | Just c' <- connOf op- , sameConn conn c'- , Just Refl <- testEquality (typeRep @c) (typeRep @a)- , Just Refl <- testEquality (typeRep @b) (typeRep @a) =- eqExpr x ya || eqExpr x yb-absorbs _ _ _ = False--sameConn :: Conn -> Conn -> Bool-sameConn ConnAnd ConnAnd = True-sameConn ConnOr ConnOr = True-sameConn _ _ = False--data Cmp = CLt | CLeq | CGt | CGeq | CEq | CNeq deriving (Eq)--data NullK = Total | FalseOnNull | UnknownOnNull deriving (Eq)--data Atom = Atom- { aCmp :: Cmp- , aThr :: !Double- , aKey :: String- , aNull :: NullK- , aIntegral :: Bool- }--cmpOf :: forall op c b r. (BinaryOp op) => op c b r -> Maybe Cmp-cmpOf _- | Just HRefl <- eqTypeRep (typeRep @op) (typeRep @NullLt) = Just CLt- | Just HRefl <- eqTypeRep (typeRep @op) (typeRep @NullLeq) = Just CLeq- | Just HRefl <- eqTypeRep (typeRep @op) (typeRep @NullGt) = Just CGt- | Just HRefl <- eqTypeRep (typeRep @op) (typeRep @NullGeq) = Just CGeq- | Just HRefl <- eqTypeRep (typeRep @op) (typeRep @NullEq) = Just CEq- | Just HRefl <- eqTypeRep (typeRep @op) (typeRep @NullNeq) = Just CNeq- | otherwise = Nothing--isLower, isUpper :: Cmp -> Bool-isLower c = c == CGt || c == CGeq-isUpper c = c == CLt || c == CLeq---- | True if @x@ is a @Maybe _@ type.-isMaybeTy :: forall x. (Columnable x) => Bool-isMaybeTy = case typeRep @x of- App con _ -> case eqTypeRep con (typeRep @Maybe) of Just HRefl -> True; _ -> False- _ -> False--litDouble :: forall b. (Columnable b) => Expr b -> Maybe Double-litDouble (Lit v) =- case testEquality (typeRep @b) (typeRep @Double) of- Just Refl -> Just v- Nothing -> case testEquality (typeRep @b) (typeRep @Int) of- Just Refl -> Just (fromIntegral v)- Nothing -> case testEquality (typeRep @b) (typeRep @(Maybe Double)) of- Just Refl -> v- Nothing -> case testEquality (typeRep @b) (typeRep @(Maybe Int)) of- Just Refl -> fromIntegral <$> v- Nothing -> Nothing-litDouble _ = Nothing--{- | True for a column lifted from an integral type (never NaN): @toDouble (col …)@-or a column whose type is itself integral.--}-integralColE :: forall c. (Columnable c) => Expr c -> Bool-integralColE (Unary op _) = unaryName op == "toDouble"-integralColE _ =- or- [ matches @Int- , matches @(Maybe Int)- ]- where- matches :: forall t. (Columnable t) => Bool- matches = case testEquality (typeRep @c) (typeRep @t) of Just Refl -> True; _ -> False--atomOf :: forall a. (Columnable a) => Expr a -> Maybe Atom-atomOf (Unary fm (Binary (op :: op c b r) (colE :: Expr c) litE))- | unaryName fm == "fromMaybe"- , Just cmp <- cmpOf op- , Just t <- litDouble litE =- Just (Atom cmp t (show (normalize colE)) FalseOnNull (integralColE colE))-atomOf (Binary (op :: op c b a) (colE :: Expr c) litE)- | Just cmp <- cmpOf op- , Just t <- litDouble litE =- let nk = if isMaybeTy @c then UnknownOnNull else Total- in Just (Atom cmp t (show (normalize colE)) nk (integralColE colE))-atomOf _ = Nothing--simplifyPredicatePair ::- forall a. (Columnable a) => Bool -> Expr a -> Expr a -> Maybe (Expr a)-simplifyPredicatePair isAnd a b = do- atomA <- atomOf a- atomB <- atomOf b- guard (aKey atomA == aKey atomB)- let nk = aNull atomA- integral = aIntegral atomA- if isAnd- then andAtoms a atomA b atomB nk integral- else orAtoms a atomA b atomB nk integral---- | Contradiction folds to a literal False unless null-rows make it unknown.-litFalseGated :: (Columnable a) => NullK -> Maybe (Expr a)-litFalseGated UnknownOnNull = Nothing-litFalseGated _ = litBoolish False--{- | Tautology to literal True is sound only for total (never-null) atoms; the-exhaustive-cover form additionally needs a non-NaN (integral) column.--}-litTrueTotal :: (Columnable a) => NullK -> Maybe (Expr a)-litTrueTotal Total = litBoolish True-litTrueTotal _ = Nothing--andAtoms ::- (Columnable a) =>- Expr a -> Atom -> Expr a -> Atom -> NullK -> Bool -> Maybe (Expr a)-andAtoms a atomA b atomB nk _ =- let cA = aCmp atomA; tA = aThr atomA; cB = aCmp atomB; tB = aThr atomB- in if- | isLower cA, isLower cB, cA == cB -> Just (if tA >= tB then a else b)- | isUpper cA, isUpper cB, cA == cB -> Just (if tA <= tB then a else b)- | isLower cA, isUpper cB -> lu cA tA cB tB- | isUpper cA, isLower cB -> lu cB tB cA tA- | cA == CEq, cB == CEq -> if tA == tB then Just a else litFalseGated nk- | cA == CEq, cB == CNeq -> if tA == tB then litFalseGated nk else Just a- | cA == CNeq, cB == CEq -> if tA == tB then litFalseGated nk else Just b- | cA == CEq -> if satisfies tA cB tB then Just a else litFalseGated nk- | cB == CEq -> if satisfies tB cA tA then Just b else litFalseGated nk- | cA == CNeq, cB == CNeq -> Nothing- | cA == CNeq -> if outside tA cB tB then Just b else Nothing- | cB == CNeq -> if outside tB cA tA then Just a else Nothing- | otherwise -> Nothing- where- lu lc lo uc hi- | lo > hi = litFalseGated nk- | lo == hi, lc == CGeq, uc == CLeq = pointEq a lo- | lo == hi = litFalseGated nk- | otherwise = Nothing--orAtoms ::- (Columnable a) =>- Expr a -> Atom -> Expr a -> Atom -> NullK -> Bool -> Maybe (Expr a)-orAtoms a atomA b atomB nk integral =- let cA = aCmp atomA; tA = aThr atomA; cB = aCmp atomB; tB = aThr atomB- in if- | isLower cA, isLower cB, cA == cB -> Just (if tA <= tB then a else b)- | isUpper cA, isUpper cB, cA == cB -> Just (if tA >= tB then a else b)- | isUpper cA- , isLower cB- , nk == Total- , integral- , covers cB tB cA tA ->- litTrueTotal nk- | isLower cA- , isUpper cB- , nk == Total- , integral- , covers cA tA cB tB ->- litTrueTotal nk- | cA == CNeq, cB == CNeq -> if tA == tB then Just a else litTrueTotal nk- | cA == CEq, cB == CNeq -> if tA == tB then litTrueTotal nk else Just b- | cA == CNeq, cB == CEq -> if tA == tB then litTrueTotal nk else Just a- | cA == CEq, cB == CEq -> if tA == tB then Just a else Nothing- | otherwise -> Nothing--{- | Build @col == t@ for the point-collapse rule; only strict @Expr Bool@ over a-@Double@ column (otherwise bail).--}-pointEq :: forall a. (Columnable a) => Expr a -> Double -> Maybe (Expr a)-pointEq atom lo = case testEquality (typeRep @a) (typeRep @Bool) of- Just Refl -> (\colE -> colE .==. Lit lo) <$> recoverColD atom- Nothing -> Nothing--recoverColD :: Expr x -> Maybe (Expr Double)-recoverColD (Binary _ (colE :: Expr c) _) =- case testEquality (typeRep @c) (typeRep @Double) of- Just Refl -> Just colE- _ -> Nothing-recoverColD (Unary _ inner) = recoverColD inner-recoverColD _ = Nothing--covers :: Cmp -> Double -> Cmp -> Double -> Bool-covers lowerCmp lo upperCmp hi =- lo < hi || (lo == hi && (lowerCmp == CGeq || upperCmp == CLeq))--satisfies :: Double -> Cmp -> Double -> Bool-satisfies t CGt tb = t > tb-satisfies t CGeq tb = t >= tb-satisfies t CLt tb = t < tb-satisfies t CLeq tb = t <= tb-satisfies _ _ _ = False--outside :: Double -> Cmp -> Double -> Bool-outside t CGt tb = t <= tb-outside t CGeq tb = t < tb-outside t CLt tb = t >= tb-outside t CLeq tb = t > tb-outside _ _ _ = False---- ------------------------------------------------------------------------------ Path-condition entailment for fitted-tree pruning.--- ------------------------------------------------------------------------------- | A known same-column threshold fact accumulated along a tree path.-data PredFact = PredFact !String !Cmp !Double---- | The fact a branch's true edge establishes (the condition holds).-factTrue :: Expr Bool -> Maybe PredFact-factTrue e = (\a -> PredFact (aKey a) (aCmp a) (aThr a)) <$> atomOf e--{- | The fact a branch's false edge establishes (the negated condition). Only-sound for non-NaN (integral) columns — a NaN row takes the false edge too,-so @¬(x>t)@ is not a clean @x<=t@ bound for floats.--}-factFalse :: Expr Bool -> Maybe PredFact-factFalse e = do- a <- atomOf e- guard (aIntegral a && aNull a == Total)- nc <- negCmp (aCmp a)- pure (PredFact (aKey a) nc (aThr a))--negCmp :: Cmp -> Maybe Cmp-negCmp CLt = Just CGeq-negCmp CLeq = Just CGt-negCmp CGt = Just CLeq-negCmp CGeq = Just CLt-negCmp _ = Nothing--{- | @entails facts cond@: 'Just' 'True' when the path facts force @cond@ true,-'Just' 'False' when they force it false, 'Nothing' when undecided.--}-entails :: [PredFact] -> Expr Bool -> Maybe Bool-entails facts cond = do- a <- atomOf cond- let decisions =- [ d- | PredFact fk fc ft <- facts- , fk == aKey a- , Just d <- [factImplies (fc, ft) (aCmp a, aThr a)]- ]- case decisions of- (d : _) -> Just d- [] -> Nothing--{- | Does the fact's solution set sit inside @cond@ ('Just' 'True'), disjoint-from it ('Just' 'False'), or neither ('Nothing')? Boundary strictness is-honoured: e.g. @x<=t@ does NOT entail @x<t@, and @x>=t ∧ x<=t@ is not empty.--}-factImplies :: (Cmp, Double) -> (Cmp, Double) -> Maybe Bool-factImplies (fc, ft) (cc, tc)- | isLower fc, isLower cc, subset = Just True- | isUpper fc, isUpper cc, subset = Just True- | isLower fc, isUpper cc, disjointAtEq = Just False- | isUpper fc, isLower cc, disjointBelow = Just False- | otherwise = Nothing- where- fIncl = fc == CGeq || fc == CLeq- cIncl = cc == CGeq || cc == CLeq- subset =- (if isLower fc then ft > tc else ft < tc)- || (ft == tc && (not fIncl || cIncl))- disjointAtEq = ft > tc || (ft == tc && not (fIncl && cIncl))- disjointBelow = ft < tc || (ft == tc && not (fIncl && cIncl))
− src-internal/DataFrame/Internal/Types.hs
@@ -1,161 +0,0 @@-{-# LANGUAGE AllowAmbiguousTypes #-}-{-# LANGUAGE ConstraintKinds #-}-{-# LANGUAGE DataKinds #-}-{-# LANGUAGE DeriveTraversable #-}-{-# LANGUAGE ExistentialQuantification #-}-{-# LANGUAGE FlexibleContexts #-}-{-# LANGUAGE GADTs #-}-{-# LANGUAGE PolyKinds #-}-{-# LANGUAGE RankNTypes #-}-{-# LANGUAGE ScopedTypeVariables #-}-{-# LANGUAGE TypeApplications #-}-{-# LANGUAGE TypeFamilies #-}-{-# LANGUAGE UndecidableInstances #-}--module DataFrame.Internal.Types where--import Data.Int (Int16, Int32, Int64, Int8)-import Data.Kind (Constraint, Type)-import Data.Typeable (Typeable)-import qualified Data.Vector.Unboxed as VU-import Data.Word (Word16, Word32, Word64, Word8)--type Columnable' a = (Typeable a, Show a, Eq a)--{- | Inline replacement for @Data.These.These@ to keep @dataframe-core@ free-of the @these@ package dependency. Only the three constructors and the-derived classes are used internally.--}-data These a b = This a | That b | These a b- deriving (Eq, Ord, Show, Read, Functor, Foldable, Traversable)--{- | A type with column representations used to select the-"right" representation when specializing the `toColumn` function.--}-data Rep- = RBoxed- | RUnboxed- | RNullableBoxed---- | Type-level if statement.-type family If (cond :: Bool) (yes :: k) (no :: k) :: k where- If 'True yes _ = yes- If 'False _ no = no---- | All unboxable types (according to the `vector` package).-type family Unboxable (a :: Type) :: Bool where- Unboxable Int = 'True- Unboxable Int8 = 'True- Unboxable Int16 = 'True- Unboxable Int32 = 'True- Unboxable Int64 = 'True- Unboxable Word = 'True- Unboxable Word8 = 'True- Unboxable Word16 = 'True- Unboxable Word32 = 'True- Unboxable Word64 = 'True- Unboxable Char = 'True- Unboxable Bool = 'True- Unboxable Double = 'True- Unboxable Float = 'True- Unboxable _ = 'False--type family Numeric (a :: Type) :: Bool where- Numeric Integer = 'True- Numeric Int = 'True- Numeric Int8 = 'True- Numeric Int16 = 'True- Numeric Int32 = 'True- Numeric Int64 = 'True- Numeric Word = 'True- Numeric Word8 = 'True- Numeric Word16 = 'True- Numeric Word32 = 'True- Numeric Word64 = 'True- Numeric Double = 'True- Numeric Float = 'True- Numeric _ = 'False---- | Compute the column representation tag for any 'a'.-type family KindOf a :: Rep where- KindOf (Maybe a) = 'RNullableBoxed- KindOf a = If (Unboxable a) 'RUnboxed 'RBoxed---- | Type-level boolean for constraint/type comparison.-data SBool (b :: Bool) where- STrue :: SBool 'True- SFalse :: SBool 'False---- | The runtime witness for our type-level branching.-class SBoolI (b :: Bool) where- sbool :: SBool b--instance SBoolI 'True where sbool = STrue-instance SBoolI 'False where sbool = SFalse---- | Runtime witness for whether @a@ is unboxable.-sUnbox :: forall a. (SBoolI (Unboxable a)) => SBool (Unboxable a)-sUnbox = sbool @(Unboxable a)--sNumeric :: forall a. (SBoolI (Numeric a)) => SBool (Numeric a)-sNumeric = sbool @(Numeric a)--type family When (flag :: Bool) (c :: Constraint) :: Constraint where- When 'True c = c- When 'False c = ()--type UnboxIf a = When (Unboxable a) (VU.Unbox a)--type family IntegralTypes (a :: Type) :: Bool where- IntegralTypes Integer = 'True- IntegralTypes Int = 'True- IntegralTypes Int8 = 'True- IntegralTypes Int16 = 'True- IntegralTypes Int32 = 'True- IntegralTypes Int64 = 'True- IntegralTypes Word = 'True- IntegralTypes Word8 = 'True- IntegralTypes Word16 = 'True- IntegralTypes Word32 = 'True- IntegralTypes Word64 = 'True- IntegralTypes _ = 'False--sIntegral :: forall a. (SBoolI (IntegralTypes a)) => SBool (IntegralTypes a)-sIntegral = sbool @(IntegralTypes a)--type IntegralIf a = When (IntegralTypes a) (Integral a)--type family FloatingTypes (a :: Type) :: Bool where- FloatingTypes Float = 'True- FloatingTypes Double = 'True- FloatingTypes _ = 'False--sFloating :: forall a. (SBoolI (FloatingTypes a)) => SBool (FloatingTypes a)-sFloating = sbool @(FloatingTypes a)--type FloatingIf a = When (FloatingTypes a) (Real a, Fractional a)--{- | Numeric type promotion: resolves the common type for mixed arithmetic.-Double dominates over Float/Int; Float dominates over Int; same types stay unchanged.--}-type family Promote (a :: Type) (b :: Type) :: Type where- Promote a a = a- Promote Double _ = Double- Promote _ Double = Double- Promote Float _ = Float- Promote _ Float = Float- Promote Int64 _ = Int64- Promote _ Int64 = Int64- Promote Int32 _ = Int32- Promote _ Int32 = Int32- Promote a _ = a--{- | Like 'Promote', but integral × integral → Double for use with './' .-Double\/Float still dominate; any two integral types (same or mixed) become Double.--}-type family PromoteDiv (a :: Type) (b :: Type) :: Type where- PromoteDiv Double _ = Double- PromoteDiv _ Double = Double- PromoteDiv Float _ = Float- PromoteDiv _ Float = Float- PromoteDiv _ _ = Double
− src-internal/DataFrame/Internal/Utf8.hs
@@ -1,95 +0,0 @@-{-# LANGUAGE BangPatterns #-}--{- | UTF-8 validation and @decodeUtf8Lenient@-parity slice decoding used by-'DataFrame.Internal.ColumnBuilder' to turn shared byte buffers into 'Text'.--}-module DataFrame.Internal.Utf8 (- isValidUtf8Slice,- isUtf8Boundary,- lenientDecodeSlice,- sliceTextVector,-) where--import qualified Data.Text as T-import qualified Data.Text.Array as A-import qualified Data.Vector as VB-import qualified Data.Vector.Mutable as VBM-import qualified Data.Vector.Unboxed as VU--import Data.Text.Internal (Text (..))-import Data.Text.Internal.Encoding.Utf8 (- DecoderResult (..),- utf8DecodeContinue,- utf8DecodeStart,- )-import Data.Text.Internal.Validate (isValidUtf8ByteArray)-import Data.Word (Word8)---- | Whether @len@ bytes starting at @off@ are well-formed UTF-8.-isValidUtf8Slice :: A.Array -> Int -> Int -> Bool-isValidUtf8Slice = isValidUtf8ByteArray-{-# INLINE isValidUtf8Slice #-}--{- | Whether a byte may start a code point (i.e. is not a continuation-byte). Field slices of a valid buffer are themselves valid iff every-field starts on a boundary.--}-isUtf8Boundary :: Word8 -> Bool-isUtf8Boundary w = w < 0x80 || w >= 0xC0-{-# INLINE isUtf8Boundary #-}--{- | Decode a byte slice exactly like @decodeUtf8Lenient@: greedy decode at-each position; any byte that cannot begin a complete, valid sequence within-the slice becomes one U+FFFD and decoding resumes at the next byte.--}-lenientDecodeSlice :: A.Array -> Int -> Int -> T.Text-lenientDecodeSlice arr off len = T.pack (go off)- where- !end = off + len- go !i- | i >= end = []- | otherwise = case tryDecode i of- Just (c, i') -> c : go i'- Nothing -> '\xFFFD' : go (i + 1)- tryDecode !i = loop (utf8DecodeStart (A.unsafeIndex arr i)) (i + 1)- where- loop (Accept c) !j = Just (c, j)- loop Reject _ = Nothing- loop (Incomplete st cp) !j- | j >= end = Nothing- | otherwise = loop (utf8DecodeContinue (A.unsafeIndex arr j) st cp) (j + 1)--{- | Slice forced 'Text' values off a shared array; row @i@ spans bytes-@[offs!i, offs!(i+1))@. Fast path validates the whole span once when every field-starts on a code-point boundary; else per-field validation with lenient decode.--}-sliceTextVector :: A.Array -> VU.Vector Int -> VB.Vector T.Text-sliceTextVector arr offs = VB.create $ do- mv <- VBM.unsafeNew n- let fill dec = go 0- where- go !i- | i >= n = pure ()- | otherwise = do- let o = VU.unsafeIndex offs i- !t = dec o (VU.unsafeIndex offs (i + 1) - o)- VBM.unsafeWrite mv i t- go (i + 1)- if fast then fill mkSlice else fill decodeField- pure mv- where- n = VU.length offs - 1- base = VU.unsafeIndex offs 0- used = VU.unsafeIndex offs n- boundariesOk !i- | i >= n = True- | otherwise =- let o = VU.unsafeIndex offs i- in (o >= used || isUtf8Boundary (A.unsafeIndex arr o))- && boundariesOk (i + 1)- fast = isValidUtf8Slice arr base (used - base) && boundariesOk 0- mkSlice o l = if l == 0 then T.empty else Text arr o l- decodeField o l- | l == 0 = T.empty- | isValidUtf8Slice arr o l = Text arr o l- | otherwise = lenientDecodeSlice arr o l
− src-internal/DataFrame/Operators.hs
@@ -1,425 +0,0 @@-{-# LANGUAGE FlexibleContexts #-}-{-# LANGUAGE GADTs #-}-{-# LANGUAGE OverloadedStrings #-}-{-# LANGUAGE TypeFamilies #-}-{-# LANGUAGE TypeOperators #-}--module DataFrame.Operators where--import Data.Function ((&))-import qualified Data.Text as T-import DataFrame.Internal.Column (Columnable)-import DataFrame.Internal.Expression (- BinUDF (MkBinaryOp),- BinaryOp (- binaryCommutative,- binaryFn,- binaryName,- binaryPrecedence,- binarySymbol- ),- Expr (Binary, Col, If, Lit, Unary),- NamedExpr,- UExpr (UExpr),- UnUDF (MkUnaryOp),- )-import DataFrame.Internal.Nullable (- BaseType,- DivWidenOp,- NullCmpResult,- NullLift2Op (applyNull2),- NullableCmpOp (nullCmpOp),- NumericWidenOp,- WidenResult,- WidenResultDiv,- divArithOp,- widenArithOp,- widenCmpOp,- )-import DataFrame.Internal.Types (Promote, PromoteDiv)--infixr 8 .^^, .^^., .^, .^.-infixl 7 .*, ./, .*., ./.-infixl 6 .+, .-, .+., .-.-infix 4 .==, .==., .<, .<., .<=, .<=., .>=, .>=., .>, .>., ./=, ./=.-infixr 3 .&&, .&&.-infixr 2 .||, .||.-infixr 0 .=--(|>) :: a -> (a -> b) -> b-(|>) = (&)--as :: (Columnable a) => Expr a -> T.Text -> NamedExpr-as expr colName = (colName, UExpr expr)--name :: (Show a) => Expr a -> T.Text-name (Col n) = n-name other =- error $- "You must call `name` on a column reference. Not the expression: " ++ show other--col :: (Columnable a) => T.Text -> Expr a-col = Col--ifThenElse :: (Columnable a) => Expr Bool -> Expr a -> Expr a -> Expr a-ifThenElse = If--lit :: (Columnable a) => a -> Expr a-lit = Lit--(.=) :: (Columnable a) => T.Text -> Expr a -> NamedExpr-(.=) = flip as--liftDecorated ::- (Columnable a, Columnable b) =>- (a -> b) -> T.Text -> Maybe T.Text -> Expr a -> Expr b-liftDecorated f opName rep = Unary (MkUnaryOp f opName rep)--lift2Decorated ::- (Columnable c, Columnable b, Columnable a) =>- (c -> b -> a) ->- T.Text ->- Maybe T.Text ->- Bool ->- Int ->- Expr c ->- Expr b ->- Expr a-lift2Decorated f opName rep comm prec =- Binary (MkBinaryOp f opName rep comm prec)--data NullEq a b c where- NullEq ::- ( NumericWidenOp (BaseType a) (BaseType b)- , NullLift2Op a b Bool (NullCmpResult a b)- , Eq (Promote (BaseType a) (BaseType b))- ) =>- NullEq a b (NullCmpResult a b)--data NullNeq a b c where- NullNeq ::- ( NumericWidenOp (BaseType a) (BaseType b)- , NullLift2Op a b Bool (NullCmpResult a b)- , Eq (Promote (BaseType a) (BaseType b))- ) =>- NullNeq a b (NullCmpResult a b)--data NullLt a b c where- NullLt ::- ( NumericWidenOp (BaseType a) (BaseType b)- , NullLift2Op a b Bool (NullCmpResult a b)- , Ord (Promote (BaseType a) (BaseType b))- ) =>- NullLt a b (NullCmpResult a b)--data NullGt a b c where- NullGt ::- ( NumericWidenOp (BaseType a) (BaseType b)- , NullLift2Op a b Bool (NullCmpResult a b)- , Ord (Promote (BaseType a) (BaseType b))- ) =>- NullGt a b (NullCmpResult a b)--data NullLeq a b c where- NullLeq ::- ( NumericWidenOp (BaseType a) (BaseType b)- , NullLift2Op a b Bool (NullCmpResult a b)- , Ord (Promote (BaseType a) (BaseType b))- ) =>- NullLeq a b (NullCmpResult a b)--data NullGeq a b c where- NullGeq ::- ( NumericWidenOp (BaseType a) (BaseType b)- , NullLift2Op a b Bool (NullCmpResult a b)- , Ord (Promote (BaseType a) (BaseType b))- ) =>- NullGeq a b (NullCmpResult a b)--data NullAnd a b c where- NullAnd ::- (NullableCmpOp a b (NullCmpResult a b), BaseType a ~ Bool) =>- NullAnd a b (NullCmpResult a b)--data NullOr a b c where- NullOr ::- (NullableCmpOp a b (NullCmpResult a b), BaseType a ~ Bool) =>- NullOr a b (NullCmpResult a b)--instance BinaryOp NullEq where- binaryFn NullEq = applyNull2 (widenCmpOp (==))- binaryName NullEq = "eq"- binarySymbol NullEq = Just ".=="- binaryCommutative NullEq = True- binaryPrecedence NullEq = 4-instance BinaryOp NullNeq where- binaryFn NullNeq = applyNull2 (widenCmpOp (/=))- binaryName NullNeq = "neq"- binarySymbol NullNeq = Just "./="- binaryCommutative NullNeq = True- binaryPrecedence NullNeq = 4-instance BinaryOp NullLt where- binaryFn NullLt = applyNull2 (widenCmpOp (<))- binaryName NullLt = "lt"- binarySymbol NullLt = Just ".<"- binaryPrecedence NullLt = 4-instance BinaryOp NullGt where- binaryFn NullGt = applyNull2 (widenCmpOp (>))- binaryName NullGt = "gt"- binarySymbol NullGt = Just ".>"- binaryPrecedence NullGt = 4-instance BinaryOp NullLeq where- binaryFn NullLeq = applyNull2 (widenCmpOp (<=))- binaryName NullLeq = "leq"- binarySymbol NullLeq = Just ".<="- binaryPrecedence NullLeq = 4-instance BinaryOp NullGeq where- binaryFn NullGeq = applyNull2 (widenCmpOp (>=))- binaryName NullGeq = "geq"- binarySymbol NullGeq = Just ".>="- binaryPrecedence NullGeq = 4-instance BinaryOp NullAnd where- binaryFn NullAnd = nullCmpOp (&&)- binaryName NullAnd = "nulland"- binarySymbol NullAnd = Just ".&&"- binaryCommutative NullAnd = True- binaryPrecedence NullAnd = 3-instance BinaryOp NullOr where- binaryFn NullOr = nullCmpOp (||)- binaryName NullOr = "nullor"- binarySymbol NullOr = Just ".||"- binaryCommutative NullOr = True- binaryPrecedence NullOr = 2--(.==.) ::- (Columnable a, Eq a) =>- Expr a ->- Expr a ->- Expr Bool-(.==.) = lift2Decorated (==) "eq" (Just ".==.") True 4--(./=.) ::- (Columnable a, Eq a) =>- Expr a ->- Expr a ->- Expr Bool-(./=.) = lift2Decorated (/=) "neq" (Just "./=.") True 4--(.<.) ::- (Columnable a, Ord a) =>- Expr a ->- Expr a ->- Expr Bool-(.<.) = lift2Decorated (<) "lt" (Just ".<.") False 4--(.>.) ::- (Columnable a, Ord a) =>- Expr a ->- Expr a ->- Expr Bool-(.>.) = lift2Decorated (>) "gt" (Just ".>.") False 4--(.<=.) ::- (Columnable a, Ord a) =>- Expr a ->- Expr a ->- Expr Bool-(.<=.) = lift2Decorated (<=) "leq" (Just ".<=.") False 4--(.>=.) ::- (Columnable a, Ord a) =>- Expr a ->- Expr a ->- Expr Bool-(.>=.) = lift2Decorated (>=) "geq" (Just ".>=.") False 4--(.+.) :: (Columnable a, Num a) => Expr a -> Expr a -> Expr a-(.+.) = (+)--(.-.) :: (Columnable a, Num a) => Expr a -> Expr a -> Expr a-(.-.) = (-)--(.*.) :: (Columnable a, Num a) => Expr a -> Expr a -> Expr a-(.*.) = (*)--(./.) :: (Columnable a, Fractional a) => Expr a -> Expr a -> Expr a-(./.) = (/)---- Nullable-aware arithmetic operators--{- | Nullable-aware addition. Works for all combinations of nullable\/non-nullable operands.-@col \@Int "x" .+ col \@(Maybe Int) "y" -- :: Expr (Maybe Int)@--}-(.+) ::- ( NumericWidenOp (BaseType a) (BaseType b)- , NullLift2Op a b (Promote (BaseType a) (BaseType b)) (WidenResult a b)- , Num (Promote (BaseType a) (BaseType b))- ) =>- Expr a ->- Expr b ->- Expr (WidenResult a b)-(.+) = lift2Decorated (applyNull2 (widenArithOp (+))) "nulladd" (Just ".+") True 6---- | Nullable-aware subtraction.-(.-) ::- ( NumericWidenOp (BaseType a) (BaseType b)- , NullLift2Op a b (Promote (BaseType a) (BaseType b)) (WidenResult a b)- , Num (Promote (BaseType a) (BaseType b))- ) =>- Expr a ->- Expr b ->- Expr (WidenResult a b)-(.-) = lift2Decorated (applyNull2 (widenArithOp (-))) "nullsub" (Just ".-") False 6---- | Nullable-aware multiplication.-(.*) ::- ( NumericWidenOp (BaseType a) (BaseType b)- , NullLift2Op a b (Promote (BaseType a) (BaseType b)) (WidenResult a b)- , Num (Promote (BaseType a) (BaseType b))- ) =>- Expr a ->- Expr b ->- Expr (WidenResult a b)-(.*) = lift2Decorated (applyNull2 (widenArithOp (*))) "nullmul" (Just ".*") True 7---- | Nullable-aware division. Integral operands are promoted to Double.-(./) ::- ( DivWidenOp (BaseType a) (BaseType b)- , NullLift2Op a b (PromoteDiv (BaseType a) (BaseType b)) (WidenResultDiv a b)- , Fractional (PromoteDiv (BaseType a) (BaseType b))- ) =>- Expr a ->- Expr b ->- Expr (WidenResultDiv a b)-(./) = lift2Decorated (applyNull2 (divArithOp (/))) "nulldiv" (Just "./") False 7---- Nullable-aware comparison operators (three-valued logic: Nothing if either operand is Nothing)--{- | Nullable-aware equality. Widens numeric operands to their common type,-so @Expr Double .== Expr Int@ typechecks. Returns @Maybe Bool@ when either-operand is nullable.--}-(.==) ::- ( NumericWidenOp (BaseType a) (BaseType b)- , NullLift2Op a b Bool (NullCmpResult a b)- , Eq (Promote (BaseType a) (BaseType b))- ) =>- Expr a ->- Expr b ->- Expr (NullCmpResult a b)-(.==) = Binary NullEq---- | Nullable-aware inequality. Widens numeric operands to their common type.-(./=) ::- ( NumericWidenOp (BaseType a) (BaseType b)- , NullLift2Op a b Bool (NullCmpResult a b)- , Eq (Promote (BaseType a) (BaseType b))- ) =>- Expr a ->- Expr b ->- Expr (NullCmpResult a b)-(./=) = Binary NullNeq---- | Nullable-aware less-than. Widens numeric operands to their common type.-(.<) ::- ( NumericWidenOp (BaseType a) (BaseType b)- , NullLift2Op a b Bool (NullCmpResult a b)- , Ord (Promote (BaseType a) (BaseType b))- ) =>- Expr a ->- Expr b ->- Expr (NullCmpResult a b)-(.<) = Binary NullLt---- | Nullable-aware greater-than. Widens numeric operands to their common type.-(.>) ::- ( NumericWidenOp (BaseType a) (BaseType b)- , NullLift2Op a b Bool (NullCmpResult a b)- , Ord (Promote (BaseType a) (BaseType b))- ) =>- Expr a ->- Expr b ->- Expr (NullCmpResult a b)-(.>) = Binary NullGt--{- | Nullable-aware less-than-or-equal. Widens numeric operands to their-common type, so @Expr Double .<= Expr Int@ typechecks.--}-(.<=) ::- ( NumericWidenOp (BaseType a) (BaseType b)- , NullLift2Op a b Bool (NullCmpResult a b)- , Ord (Promote (BaseType a) (BaseType b))- ) =>- Expr a ->- Expr b ->- Expr (NullCmpResult a b)-(.<=) = Binary NullLeq---- | Nullable-aware greater-than-or-equal. Widens numeric operands to their common type.-(.>=) ::- ( NumericWidenOp (BaseType a) (BaseType b)- , NullLift2Op a b Bool (NullCmpResult a b)- , Ord (Promote (BaseType a) (BaseType b))- ) =>- Expr a ->- Expr b ->- Expr (NullCmpResult a b)-(.>=) = Binary NullGeq--(.&&.) :: Expr Bool -> Expr Bool -> Expr Bool-(.&&.) = lift2Decorated (&&) "and" (Just ".&&.") True 3--(.||.) :: Expr Bool -> Expr Bool -> Expr Bool-(.||.) = lift2Decorated (||) "or" (Just ".||.") True 2---- | Nullable-aware logical AND. Returns @Maybe Bool@ when either operand is nullable.-(.&&) ::- (NullableCmpOp a b (NullCmpResult a b), BaseType a ~ Bool) =>- Expr a ->- Expr b ->- Expr (NullCmpResult a b)-(.&&) = Binary NullAnd---- | Nullable-aware logical OR. Returns @Maybe Bool@ when either operand is nullable.-(.||) ::- (NullableCmpOp a b (NullCmpResult a b), BaseType a ~ Bool) =>- Expr a ->- Expr b ->- Expr (NullCmpResult a b)-(.||) = Binary NullOr--(.^^) ::- ( Columnable (BaseType a)- , Columnable (BaseType b)- , Fractional (BaseType a)- , Integral (BaseType b)- , NumericWidenOp (BaseType a) (BaseType b)- , NullLift2Op a b (BaseType a) a- , Num (Promote (BaseType a) (BaseType b))- ) =>- Expr a -> Expr b -> Expr a-(.^^) = lift2Decorated (applyNull2 (^^)) "pow" (Just ".^^") False 8--(.^) ::- ( Columnable (BaseType a)- , Columnable (BaseType b)- , Num (BaseType a)- , Integral (BaseType b)- , NumericWidenOp (BaseType a) (BaseType b)- , NullLift2Op a b (BaseType a) a- , Num (Promote (BaseType a) (BaseType b))- ) =>- Expr a -> Expr b -> Expr a-(.^) = lift2Decorated (applyNull2 (^)) "pow" (Just ".^") False 8---- Same-type (non-nullable) exponentiation operators--(.^^.) ::- (Columnable a, Columnable b, Fractional a, Integral b) =>- Expr a -> Expr b -> Expr a-(.^^.) = lift2Decorated (^^) "pow" (Just ".^^.") False 8--(.^.) ::- (Columnable a, Columnable b, Num a, Integral b) =>- Expr a -> Expr b -> Expr a-(.^.) = lift2Decorated (^) "pow" (Just ".^.") False 8
src/DataFrame/Core.hs view
@@ -72,6 +72,7 @@ toList, toVector, )+import DataFrame.Internal.Column.Types (Columnable') import DataFrame.Internal.DataFrame ( DataFrame, GroupedDataFrame,@@ -107,4 +108,3 @@ toRowList, toRowVector, )-import DataFrame.Internal.Types (Columnable')
+ src/DataFrame/Expression/Operators.hs view
@@ -0,0 +1,12 @@+{- |+The public name for dataframe's expression operators.++This is a re-export shim: the implementation lives in+"DataFrame.Internal.Expression.Operators", which is internal and may be+reorganised without notice. Depend on this module instead.+-}+module DataFrame.Expression.Operators (+ module DataFrame.Internal.Expression.Operators,+) where++import DataFrame.Internal.Expression.Operators
src/DataFrame/Typed/Schema.hs view
@@ -78,7 +78,7 @@ import Type.Reflection (SomeTypeRep, Typeable, someTypeRep) import DataFrame.Internal.Column (Columnable)-import DataFrame.Internal.Types (These)+import DataFrame.Internal.Column.Types (These) -- | Look up the element type of a column by name. type family Lookup (name :: Symbol) (cols :: [(Symbol, Type)]) :: Type where
src/DataFrame/Typed/Types.hs view
@@ -40,9 +40,9 @@ import qualified Data.Text as T import DataFrame.Internal.Column (Columnable)+import DataFrame.Internal.Column.Types (These (..)) import qualified DataFrame.Internal.DataFrame as D import DataFrame.Internal.Expression (Expr, NamedExpr, UExpr (..))-import DataFrame.Internal.Types (These (..)) {- | A phantom-typed wrapper over the untyped 'DataFrame'.