| Copyright | (c) 2024 - 2026 Michael Chavinda |
|---|---|
| License | MIT |
| Maintainer | mschavinda@gmail.com |
| Stability | experimental |
| Safe Haskell | None |
| Language | Haskell2010 |
DataFrame.Typed
Contents
- Core types
- Typed expressions
- Same-type comparison operators
- Nullable-aware arithmetic operators
- Nullable-aware comparison operators (three-valued logic)
- Logical operators
- Aggregation expression combinators
- Expression combinators (full DataFrame.Functions parity)
- Cast / coercion expressions
- Typed sort orders
- Freeze / thaw boundary
- Typed column access
- Schema-preserving operations
- Schema-modifying operations
- Metadata
- Vertical merge
- Set algebra (topos operations)
- Joins
- GroupBy and Aggregation
- Column transformations
- Sampling and splitting
- Frequencies
- Template Haskell
- Record bridge (ADT - TypedDataFrame)
- Generics opt-in for schema derivation
- Schema type families (for advanced use)
- Constraints
Description
A type-safe layer over the dataframe library.
This module provides TypedDataFrame, a phantom-typed wrapper around
the untyped DataFrame that tracks column names and types at compile time.
All operations delegate to the untyped core at runtime; the phantom type
is updated at compile time to reflect schema changes.
Key difference from untyped API: TExpr
All expression-taking operations use TExpr (typed expressions) instead
of raw Expr. Column references are validated at compile time:
{-# LANGUAGE DataKinds, TypeApplications, TypeOperators #-}
import qualified DataFrame.Typed as T
type People = '[ '("name", Text), '("age", Int)]
main = do
raw <- D.readCsv "people.csv"
case T.freeze @People raw of
Nothing -> putStrLn "Schema mismatch!"
Just df -> do
let adults = T.filterWhere (T.col @"age" T..>=. T.lit 18) df
let names = T.columnAsList @"name" adults -- :: [Text]
print names
Column references like T.col @"age" are checked at compile time — if the
column doesn't exist or has the wrong type, you get a type error, not a
runtime exception.
filterAllJust tracks Maybe-stripping
df :: TypedDataFrame '[ '("x", Maybe Double), '("y", Int)]
T.filterAllJust df :: TypedDataFrame '[ '("x", Double), '("y", Int)]
Typed aggregation
result = T.aggregate
( T.as @"total" (T.sum (T.col @"salary"))
. T.as @"count" (T.count (T.col @"salary"))
)
(T.groupBy @'["dept"] employees)
Synopsis
- data TypedDataFrame (cols :: [(Symbol, Type)])
- data TypedGrouped (keys :: [Symbol]) (cols :: [(Symbol, Type)])
- data These a b
- newtype TExpr (cols :: [(Symbol, Type)]) a = TExpr {}
- col :: forall (name :: Symbol) (cols :: [(Symbol, Type)]) a. (KnownSymbol name, a ~ SafeLookup name cols, Columnable a, AssertPresent name cols) => TExpr cols a
- lit :: forall a (cols :: [(Symbol, Type)]). Columnable a => a -> TExpr cols a
- ifThenElse :: forall a (cols :: [(Symbol, Type)]). Columnable a => TExpr cols Bool -> TExpr cols a -> TExpr cols a -> TExpr cols a
- lift :: forall a b (cols :: [(Symbol, Type)]). (Columnable a, Columnable b) => (a -> b) -> TExpr cols a -> TExpr cols b
- lift2 :: forall a b c (cols :: [(Symbol, Type)]). (Columnable a, Columnable b, Columnable c) => (a -> b -> c) -> TExpr cols a -> TExpr cols b -> TExpr cols c
- nullLift :: forall a r (cols :: [(Symbol, Type)]). (NullLift1Op a r (NullLift1Result a r), Columnable (NullLift1Result a r)) => (BaseType a -> r) -> TExpr cols a -> TExpr cols (NullLift1Result a r)
- nullLift2 :: forall a b r (cols :: [(Symbol, Type)]). (NullLift2Op a b r (NullLift2Result a b r), Columnable (NullLift2Result a b r)) => (BaseType a -> BaseType b -> r) -> TExpr cols a -> TExpr cols b -> TExpr cols (NullLift2Result a b r)
- (.==.) :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Eq a) => TExpr cols a -> TExpr cols a -> TExpr cols Bool
- (./=.) :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Eq a) => TExpr cols a -> TExpr cols a -> TExpr cols Bool
- (.<.) :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TExpr cols a -> TExpr cols Bool
- (.<=.) :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TExpr cols a -> TExpr cols Bool
- (.>=.) :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TExpr cols a -> TExpr cols Bool
- (.>.) :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TExpr cols a -> TExpr cols Bool
- (.+) :: forall a b (cols :: [(Symbol, Type)]). (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b (Promote (BaseType a) (BaseType b)) (WidenResult a b), Num (Promote (BaseType a) (BaseType b))) => TExpr cols a -> TExpr cols b -> TExpr cols (WidenResult a b)
- (.-) :: forall a b (cols :: [(Symbol, Type)]). (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b (Promote (BaseType a) (BaseType b)) (WidenResult a b), Num (Promote (BaseType a) (BaseType b))) => TExpr cols a -> TExpr cols b -> TExpr cols (WidenResult a b)
- (.*) :: forall a b (cols :: [(Symbol, Type)]). (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b (Promote (BaseType a) (BaseType b)) (WidenResult a b), Num (Promote (BaseType a) (BaseType b))) => TExpr cols a -> TExpr cols b -> TExpr cols (WidenResult a b)
- (./) :: forall a b (cols :: [(Symbol, Type)]). (DivWidenOp (BaseType a) (BaseType b), NullLift2Op a b (PromoteDiv (BaseType a) (BaseType b)) (WidenResultDiv a b), Fractional (PromoteDiv (BaseType a) (BaseType b))) => TExpr cols a -> TExpr cols b -> TExpr cols (WidenResultDiv a b)
- (.==) :: forall a b (cols :: [(Symbol, Type)]). (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Eq (Promote (BaseType a) (BaseType b))) => TExpr cols a -> TExpr cols b -> TExpr cols (NullCmpResult a b)
- (./=) :: forall a b (cols :: [(Symbol, Type)]). (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Eq (Promote (BaseType a) (BaseType b))) => TExpr cols a -> TExpr cols b -> TExpr cols (NullCmpResult a b)
- (.<) :: forall a b (cols :: [(Symbol, Type)]). (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Ord (Promote (BaseType a) (BaseType b))) => TExpr cols a -> TExpr cols b -> TExpr cols (NullCmpResult a b)
- (.<=) :: forall a b (cols :: [(Symbol, Type)]). (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Ord (Promote (BaseType a) (BaseType b))) => TExpr cols a -> TExpr cols b -> TExpr cols (NullCmpResult a b)
- (.>=) :: forall a b (cols :: [(Symbol, Type)]). (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Ord (Promote (BaseType a) (BaseType b))) => TExpr cols a -> TExpr cols b -> TExpr cols (NullCmpResult a b)
- (.>) :: forall a b (cols :: [(Symbol, Type)]). (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Ord (Promote (BaseType a) (BaseType b))) => TExpr cols a -> TExpr cols b -> TExpr cols (NullCmpResult a b)
- (.&&.) :: forall (cols :: [(Symbol, Type)]). TExpr cols Bool -> TExpr cols Bool -> TExpr cols Bool
- (.||.) :: forall (cols :: [(Symbol, Type)]). TExpr cols Bool -> TExpr cols Bool -> TExpr cols Bool
- not :: forall (cols :: [(Symbol, Type)]). TExpr cols Bool -> TExpr cols Bool
- sum :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Num a) => TExpr cols a -> TExpr cols a
- mean :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a) => TExpr cols a -> TExpr cols Double
- median :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a, Unbox a) => TExpr cols a -> TExpr cols Double
- count :: forall a (cols :: [(Symbol, Type)]). Columnable a => TExpr cols a -> TExpr cols Int
- countAll :: forall (cols :: [(Symbol, Type)]). TExpr cols Int
- minimum :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TExpr cols a
- maximum :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TExpr cols a
- collect :: forall a (cols :: [(Symbol, Type)]). Columnable a => TExpr cols a -> TExpr cols [a]
- over :: forall (names :: [Symbol]) (cols :: [(Symbol, Type)]) a. (Columnable a, AllKnownSymbol names, AssertAllPresent names cols) => TExpr cols a -> TExpr cols a
- div :: forall a (cols :: [(Symbol, Type)]). (Integral a, Columnable a) => TExpr cols a -> TExpr cols a -> TExpr cols a
- mod :: forall a (cols :: [(Symbol, Type)]). (Integral a, Columnable a) => TExpr cols a -> TExpr cols a -> TExpr cols a
- mode :: forall a (cols :: [(Symbol, Type)]). (Ord a, Columnable a, Eq a) => TExpr cols a -> TExpr cols a
- sumMaybe :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Num a) => TExpr cols (Maybe a) -> TExpr cols a
- meanMaybe :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a) => TExpr cols (Maybe a) -> TExpr cols Double
- variance :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a, Unbox a) => TExpr cols a -> TExpr cols Double
- medianMaybe :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a) => TExpr cols (Maybe a) -> TExpr cols Double
- percentile :: forall (cols :: [(Symbol, Type)]). Int -> TExpr cols Double -> TExpr cols Double
- stddev :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a, Unbox a) => TExpr cols a -> TExpr cols Double
- stddevMaybe :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a) => TExpr cols (Maybe a) -> TExpr cols Double
- zScore :: forall (cols :: [(Symbol, Type)]). TExpr cols Double -> TExpr cols Double
- pow :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Num a) => TExpr cols a -> Int -> TExpr cols a
- relu :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Num a, Ord a) => TExpr cols a -> TExpr cols a
- min :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TExpr cols a -> TExpr cols a
- max :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TExpr cols a -> TExpr cols a
- reduce :: forall a b (cols :: [(Symbol, Type)]). (Columnable a, Columnable b) => TExpr cols b -> a -> (a -> b -> a) -> TExpr cols a
- toMaybe :: forall a (cols :: [(Symbol, Type)]). Columnable a => TExpr cols a -> TExpr cols (Maybe a)
- fromMaybe :: forall a (cols :: [(Symbol, Type)]). Columnable a => a -> TExpr cols (Maybe a) -> TExpr cols a
- isJust :: forall a (cols :: [(Symbol, Type)]). Columnable a => TExpr cols (Maybe a) -> TExpr cols Bool
- isNothing :: forall a (cols :: [(Symbol, Type)]). Columnable a => TExpr cols (Maybe a) -> TExpr cols Bool
- fromJust :: forall a (cols :: [(Symbol, Type)]). Columnable a => TExpr cols (Maybe a) -> TExpr cols a
- whenPresent :: forall a b (cols :: [(Symbol, Type)]). (Columnable a, Columnable b) => (a -> b) -> TExpr cols (Maybe a) -> TExpr cols (Maybe b)
- whenBothPresent :: forall a b c (cols :: [(Symbol, Type)]). (Columnable a, Columnable b, Columnable c) => (a -> b -> c) -> TExpr cols (Maybe a) -> TExpr cols (Maybe b) -> TExpr cols (Maybe c)
- recode :: forall a b (cols :: [(Symbol, Type)]). (Columnable a, Columnable b, Show a, Show b, Show (a, b)) => [(a, b)] -> TExpr cols a -> TExpr cols (Maybe b)
- recodeWithCondition :: forall a b (cols :: [(Symbol, Type)]). (Columnable a, Columnable b) => TExpr cols b -> [(TExpr cols a -> TExpr cols Bool, b)] -> TExpr cols a -> TExpr cols b
- recodeWithDefault :: forall a b (cols :: [(Symbol, Type)]). (Columnable a, Columnable b, Show (a, b)) => b -> [(a, b)] -> TExpr cols a -> TExpr cols b
- firstOrNothing :: forall a (cols :: [(Symbol, Type)]). Columnable a => TExpr cols [a] -> TExpr cols (Maybe a)
- lastOrNothing :: forall a (cols :: [(Symbol, Type)]). Columnable a => TExpr cols [a] -> TExpr cols (Maybe a)
- splitOn :: forall (cols :: [(Symbol, Type)]). Text -> TExpr cols Text -> TExpr cols [Text]
- match :: forall (cols :: [(Symbol, Type)]). Text -> TExpr cols Text -> TExpr cols (Maybe Text)
- matchAll :: forall (cols :: [(Symbol, Type)]). Text -> TExpr cols Text -> TExpr cols [Text]
- parseDate :: forall t (cols :: [(Symbol, Type)]). (ParseTime t, Columnable t) => Text -> TExpr cols Text -> TExpr cols (Maybe t)
- daysBetween :: forall (cols :: [(Symbol, Type)]). TExpr cols Day -> TExpr cols Day -> TExpr cols Int
- bind :: forall a m b (cols :: [(Symbol, Type)]). (Columnable a, Columnable (m a), Monad m, Columnable b, Columnable (m b)) => (a -> m b) -> TExpr cols (m a) -> TExpr cols (m b)
- castExpr :: forall b (cols :: [(Symbol, Type)]) src. (Columnable b, Columnable src, Read b) => TExpr cols src -> TExpr cols (Maybe b)
- castExprWithDefault :: forall b (cols :: [(Symbol, Type)]) src. (Columnable b, Columnable src, Read b) => b -> TExpr cols src -> TExpr cols b
- castExprEither :: forall b (cols :: [(Symbol, Type)]) src. (Columnable b, Columnable src, Read b) => TExpr cols src -> TExpr cols (Either Text b)
- unsafeCastExpr :: forall b (cols :: [(Symbol, Type)]) src. (Columnable b, Columnable src, Read b) => TExpr cols src -> TExpr cols b
- toDouble :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a) => TExpr cols a -> TExpr cols Double
- data TSortOrder (cols :: [(Symbol, Type)]) where
- Asc :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TSortOrder cols
- Desc :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TSortOrder cols
- asc :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TSortOrder cols
- desc :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TSortOrder cols
- freeze :: forall (cols :: [(Symbol, Type)]). KnownSchema cols => DataFrame -> Maybe (TypedDataFrame cols)
- freezeWithError :: forall (cols :: [(Symbol, Type)]). KnownSchema cols => DataFrame -> Either Text (TypedDataFrame cols)
- thaw :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> DataFrame
- unsafeFreeze :: forall (cols :: [(Symbol, Type)]). DataFrame -> TypedDataFrame cols
- columnAsVector :: forall (name :: Symbol) (cols :: [(Symbol, Type)]) a. (KnownSymbol name, a ~ SafeLookup name cols, Columnable a, AssertPresent name cols) => TypedDataFrame cols -> Vector a
- columnAsList :: forall (name :: Symbol) (cols :: [(Symbol, Type)]) a. (KnownSymbol name, a ~ SafeLookup name cols, Columnable a, AssertPresent name cols) => TypedDataFrame cols -> [a]
- columnAsIntVector :: forall (name :: Symbol) (cols :: [(Symbol, Type)]) a. (KnownSymbol name, a ~ SafeLookup name cols, Columnable a, AssertRealColumn "columnAsIntVector" name a, Real a, Unbox a, AssertPresent name cols) => TypedDataFrame cols -> Vector Int
- columnAsDoubleVector :: forall (name :: Symbol) (cols :: [(Symbol, Type)]) a. (KnownSymbol name, a ~ SafeLookup name cols, Columnable a, AssertRealColumn "columnAsDoubleVector" name a, Real a, Unbox a, AssertPresent name cols) => TypedDataFrame cols -> Vector Double
- columnAsFloatVector :: forall (name :: Symbol) (cols :: [(Symbol, Type)]) a. (KnownSymbol name, a ~ SafeLookup name cols, Columnable a, AssertRealColumn "columnAsFloatVector" name a, Real a, Unbox a, AssertPresent name cols) => TypedDataFrame cols -> Vector Float
- columnAsUnboxedVector :: forall (name :: Symbol) (cols :: [(Symbol, Type)]) a. (KnownSymbol name, a ~ SafeLookup name cols, Columnable a, Unbox a, AssertPresent name cols) => TypedDataFrame cols -> Vector a
- toDoubleMatrix :: forall (cols :: [(Symbol, Type)]). AllColumnsReal "toDoubleMatrix" cols => TypedDataFrame cols -> Vector (Vector Double)
- toFloatMatrix :: forall (cols :: [(Symbol, Type)]). AllColumnsReal "toFloatMatrix" cols => TypedDataFrame cols -> Vector (Vector Float)
- toIntMatrix :: forall (cols :: [(Symbol, Type)]). AllColumnsReal "toIntMatrix" cols => TypedDataFrame cols -> Vector (Vector Int)
- filterWhere :: forall (cols :: [(Symbol, Type)]). TExpr cols Bool -> TypedDataFrame cols -> TypedDataFrame cols
- filter :: forall a (cols :: [(Symbol, Type)]). Columnable a => TExpr cols a -> (a -> Bool) -> TypedDataFrame cols -> TypedDataFrame cols
- filterBy :: forall a (cols :: [(Symbol, Type)]). Columnable a => (a -> Bool) -> TExpr cols a -> TypedDataFrame cols -> TypedDataFrame cols
- filterAllJust :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> TypedDataFrame (StripAllMaybe cols)
- filterJust :: forall (name :: Symbol) (cols :: [(Symbol, Type)]). (KnownSymbol name, AssertPresent name cols) => TypedDataFrame cols -> TypedDataFrame (StripMaybeAt name cols)
- filterNothing :: forall (name :: Symbol) (cols :: [(Symbol, Type)]). (KnownSymbol name, AssertPresent name cols) => TypedDataFrame cols -> TypedDataFrame cols
- filterAllNothing :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> TypedDataFrame cols
- sortBy :: forall (cols :: [(Symbol, Type)]). [TSortOrder cols] -> TypedDataFrame cols -> TypedDataFrame cols
- take :: forall (cols :: [(Symbol, Type)]). Int -> TypedDataFrame cols -> TypedDataFrame cols
- takeLast :: forall (cols :: [(Symbol, Type)]). Int -> TypedDataFrame cols -> TypedDataFrame cols
- drop :: forall (cols :: [(Symbol, Type)]). Int -> TypedDataFrame cols -> TypedDataFrame cols
- dropLast :: forall (cols :: [(Symbol, Type)]). Int -> TypedDataFrame cols -> TypedDataFrame cols
- range :: forall (cols :: [(Symbol, Type)]). (Int, Int) -> TypedDataFrame cols -> TypedDataFrame cols
- cube :: forall (cols :: [(Symbol, Type)]). (Int, Int) -> TypedDataFrame cols -> TypedDataFrame cols
- distinct :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> TypedDataFrame cols
- sample :: forall g (cols :: [(Symbol, Type)]). RandomGen g => g -> Double -> TypedDataFrame cols -> TypedDataFrame cols
- shuffle :: forall g (cols :: [(Symbol, Type)]). RandomGen g => g -> TypedDataFrame cols -> TypedDataFrame cols
- derive :: forall (name :: Symbol) a (cols :: [(Symbol, Type)]). (KnownSymbol name, Columnable a, AssertAbsent name cols) => TExpr cols a -> TypedDataFrame cols -> TypedDataFrame (Snoc cols '(name, a))
- impute :: forall (name :: Symbol) a (cols :: [(Symbol, Type)]). (KnownSymbol name, Columnable a, Maybe a ~ Lookup name cols) => a -> TypedDataFrame cols -> TypedDataFrame (Impute name cols)
- select :: forall (names :: [Symbol]) (cols :: [(Symbol, Type)]). (AllKnownSymbol names, AssertAllPresent names cols) => TypedDataFrame cols -> TypedDataFrame (SubsetSchema names cols)
- exclude :: forall (names :: [Symbol]) (cols :: [(Symbol, Type)]). AllKnownSymbol names => TypedDataFrame cols -> TypedDataFrame (ExcludeSchema names cols)
- rename :: forall (old :: Symbol) (new :: Symbol) (cols :: [(Symbol, Type)]). (KnownSymbol old, KnownSymbol new) => TypedDataFrame cols -> TypedDataFrame (RenameInSchema old new cols)
- renameMany :: forall (pairs :: [(Symbol, Symbol)]) (cols :: [(Symbol, Type)]). AllKnownPairs pairs => TypedDataFrame cols -> TypedDataFrame (RenameManyInSchema pairs cols)
- insert :: forall (name :: Symbol) a (cols :: [(Symbol, Type)]) t. (KnownSymbol name, Columnable a, Foldable t, AssertAbsent name cols) => t a -> TypedDataFrame cols -> TypedDataFrame ('(name, a) ': cols)
- insertColumn :: forall (name :: Symbol) a (cols :: [(Symbol, Type)]). (KnownSymbol name, Columnable a, AssertAbsent name cols) => Column -> TypedDataFrame cols -> TypedDataFrame ('(name, a) ': cols)
- insertVector :: forall (name :: Symbol) a (cols :: [(Symbol, Type)]). (KnownSymbol name, Columnable a, AssertAbsent name cols) => Vector a -> TypedDataFrame cols -> TypedDataFrame ('(name, a) ': cols)
- cloneColumn :: forall (old :: Symbol) (new :: Symbol) (cols :: [(Symbol, Type)]). (KnownSymbol old, KnownSymbol new, AssertPresent old cols, AssertAbsent new cols) => TypedDataFrame cols -> TypedDataFrame ('(new, Lookup old cols) ': cols)
- dropColumn :: forall (name :: Symbol) (cols :: [(Symbol, Type)]). (KnownSymbol name, AssertPresent name cols) => TypedDataFrame cols -> TypedDataFrame (RemoveColumn name cols)
- replaceColumn :: forall (name :: Symbol) a (cols :: [(Symbol, Type)]). (KnownSymbol name, Columnable a, a ~ SafeLookup name cols, AssertPresent name cols) => TExpr cols a -> TypedDataFrame cols -> TypedDataFrame cols
- dimensions :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> (Int, Int)
- nRows :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> Int
- nColumns :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> Int
- columnNames :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> [Text]
- append :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> TypedDataFrame cols -> TypedDataFrame cols
- union :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> TypedDataFrame cols -> TypedDataFrame cols
- intersect :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> TypedDataFrame cols -> TypedDataFrame cols
- difference :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> TypedDataFrame cols -> TypedDataFrame cols
- symmetricDifference :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> TypedDataFrame cols -> TypedDataFrame cols
- innerJoin :: forall (keys :: [Symbol]) (left :: [(Symbol, Type)]) (right :: [(Symbol, Type)]). (AllKnownSymbol keys, AssertAllPresent keys left, AssertAllPresent keys right, AssertKeyTypesMatch keys left right) => TypedDataFrame left -> TypedDataFrame right -> TypedDataFrame (InnerJoinSchema keys left right)
- leftJoin :: forall (keys :: [Symbol]) (left :: [(Symbol, Type)]) (right :: [(Symbol, Type)]). (AllKnownSymbol keys, AssertAllPresent keys left, AssertAllPresent keys right, AssertKeyTypesMatch keys left right) => TypedDataFrame left -> TypedDataFrame right -> TypedDataFrame (LeftJoinSchema keys left right)
- rightJoin :: forall (keys :: [Symbol]) (left :: [(Symbol, Type)]) (right :: [(Symbol, Type)]). (AllKnownSymbol keys, AssertAllPresent keys left, AssertAllPresent keys right, AssertKeyTypesMatch keys left right) => TypedDataFrame left -> TypedDataFrame right -> TypedDataFrame (RightJoinSchema keys left right)
- fullOuterJoin :: forall (keys :: [Symbol]) (left :: [(Symbol, Type)]) (right :: [(Symbol, Type)]). (AllKnownSymbol keys, AssertAllPresent keys left, AssertAllPresent keys right, AssertKeyTypesMatch keys left right) => TypedDataFrame left -> TypedDataFrame right -> TypedDataFrame (FullOuterJoinSchema keys left right)
- groupBy :: forall (keys :: [Symbol]) (cols :: [(Symbol, Type)]). (AllKnownSymbol keys, AssertAllPresent keys cols) => TypedDataFrame cols -> TypedGrouped keys cols
- as :: forall (name :: Symbol) a (keys :: [Symbol]) (cols :: [(Symbol, Type)]) (aggs :: [(Symbol, Type)]). (KnownSymbol name, Columnable a) => TExpr cols a -> TAgg keys cols aggs -> TAgg keys cols ('(name, a) ': aggs)
- aggregate :: forall (keys :: [Symbol]) (cols :: [(Symbol, Type)]) (aggs :: [(Symbol, Type)]). (TAgg keys cols ('[] :: [(Symbol, Type)]) -> TAgg keys cols aggs) -> TypedGrouped keys cols -> TypedDataFrame (Append (GroupKeyColumns keys cols) (Reverse aggs))
- aggregateUntyped :: forall (keys :: [Symbol]) (cols :: [(Symbol, Type)]). [NamedExpr] -> TypedGrouped keys cols -> DataFrame
- applyColumn :: forall (name :: Symbol) a b (cols :: [(Symbol, Type)]). (KnownSymbol name, a ~ SafeLookup name cols, Columnable a, Columnable b, AssertPresent name cols) => (a -> b) -> TypedDataFrame cols -> TypedDataFrame (SetColumnType name b cols)
- applyMany :: forall (names :: [Symbol]) a (cols :: [(Symbol, Type)]). (AllKnownSymbol names, Columnable a, AssertAllColumnsHaveType names a cols) => (a -> a) -> TypedDataFrame cols -> TypedDataFrame cols
- applyWhere :: forall (filterName :: Symbol) (targetName :: Symbol) a b (cols :: [(Symbol, Type)]). (KnownSymbol filterName, KnownSymbol targetName, a ~ SafeLookup filterName cols, b ~ SafeLookup targetName cols, Columnable a, Columnable b, AssertPresent filterName cols, AssertPresent targetName cols) => (a -> Bool) -> (b -> b) -> TypedDataFrame cols -> TypedDataFrame cols
- applyAtIndex :: forall (name :: Symbol) a (cols :: [(Symbol, Type)]). (KnownSymbol name, a ~ SafeLookup name cols, Columnable a, AssertPresent name cols) => Int -> (a -> a) -> TypedDataFrame cols -> TypedDataFrame cols
- safeApply :: forall (name :: Symbol) a b (cols :: [(Symbol, Type)]). (KnownSymbol name, a ~ SafeLookup name cols, Columnable a, Columnable b, AssertPresent name cols) => (a -> b) -> TypedDataFrame cols -> Either DataFrameException (TypedDataFrame (SetColumnType name b cols))
- deriveWithExpr :: forall (name :: Symbol) a (cols :: [(Symbol, Type)]). (KnownSymbol name, Columnable a, AssertAbsent name cols) => TExpr cols a -> TypedDataFrame cols -> (TExpr (Snoc cols '(name, a)) a, TypedDataFrame (Snoc cols '(name, a)))
- insertWithDefault :: forall (name :: Symbol) a (cols :: [(Symbol, Type)]) t. (KnownSymbol name, Columnable a, Foldable t, AssertAbsent name cols) => a -> t a -> TypedDataFrame cols -> TypedDataFrame ('(name, a) ': cols)
- insertVectorWithDefault :: forall (name :: Symbol) a (cols :: [(Symbol, Type)]). (KnownSymbol name, Columnable a, AssertAbsent name cols) => a -> Vector a -> TypedDataFrame cols -> TypedDataFrame ('(name, a) ': cols)
- insertUnboxedVector :: forall (name :: Symbol) a (cols :: [(Symbol, Type)]). (KnownSymbol name, Columnable a, Unbox a, AssertAbsent name cols) => Vector a -> TypedDataFrame cols -> TypedDataFrame ('(name, a) ': cols)
- (|||) :: forall (left :: [(Symbol, Type)]) (right :: [(Symbol, Type)]). AssertDisjoint left right => TypedDataFrame left -> TypedDataFrame right -> TypedDataFrame (Append left right)
- randomSplit :: forall g (cols :: [(Symbol, Type)]). RandomGen g => g -> Double -> TypedDataFrame cols -> (TypedDataFrame cols, TypedDataFrame cols)
- kFolds :: forall g (cols :: [(Symbol, Type)]). RandomGen g => g -> Int -> TypedDataFrame cols -> [TypedDataFrame cols]
- selectRows :: forall (cols :: [(Symbol, Type)]). [Int] -> TypedDataFrame cols -> TypedDataFrame cols
- stratifiedSample :: forall g a (cols :: [(Symbol, Type)]). (SplittableGen g, Columnable a) => g -> Double -> TExpr cols a -> TypedDataFrame cols -> TypedDataFrame cols
- stratifiedSplit :: forall g a (cols :: [(Symbol, Type)]). (SplittableGen g, Columnable a) => g -> Double -> TExpr cols a -> TypedDataFrame cols -> (TypedDataFrame cols, TypedDataFrame cols)
- valueCounts :: forall a (cols :: [(Symbol, Type)]). (Ord a, Columnable a) => TExpr cols a -> TypedDataFrame cols -> [(a, Int)]
- valueProportions :: forall a (cols :: [(Symbol, Type)]). (Ord a, Columnable a) => TExpr cols a -> TypedDataFrame cols -> [(a, Double)]
- deriveSchema :: String -> DataFrame -> DecsQ
- deriveSchemaFromCsvFile :: String -> String -> DecsQ
- deriveSchemaFromCsvFileWith :: ReadOptions -> String -> String -> DecsQ
- deriveSchemaFromParquetFile :: String -> String -> DecsQ
- deriveSchemaFromType :: Name -> DecsQ
- deriveSchemaFromTypeWith :: SchemaOptions -> Name -> DecsQ
- data SchemaOptions = SchemaOptions {}
- defaultSchemaOptions :: SchemaOptions
- class HasSchema a where
- fromRecordsTyped :: HasSchema a => [a] -> TypedDataFrame (Schema a)
- toRecordsTyped :: HasSchema a => TypedDataFrame (Schema a) -> Either Text [a]
- type SchemaOf a = RepToSchema 'SnakeCase (Rep a)
- type SchemaOfRaw a = RepToSchema 'IdentityCase (Rep a)
- data NameCase
- genericToColumns :: (Generic a, GHasColumns (Rep a)) => [a] -> [(Text, Column)]
- genericFromColumns :: (Generic a, GHasColumns (Rep a)) => DataFrame -> Either Text [a]
- type family Lookup (name :: Symbol) (cols :: [(Symbol, Type)]) where ...
- type family SafeLookup (name :: Symbol) (cols :: [(Symbol, Type)]) where ...
- type family HasName (name :: Symbol) (cols :: [(Symbol, Type)]) :: Bool where ...
- type family SubsetSchema (names :: [Symbol]) (cols :: [(Symbol, Type)]) :: [(Symbol, Type)] where ...
- type family ExcludeSchema (names :: [Symbol]) (cols :: [(Symbol, Type)]) :: [(Symbol, Type)] where ...
- type family RenameInSchema (old :: Symbol) (new :: Symbol) (cols :: [(Symbol, Type)]) :: [(Symbol, Type)] where ...
- type family RenameManyInSchema (pairs :: [(Symbol, Symbol)]) (cols :: [(Symbol, Type)]) :: [(Symbol, Type)] where ...
- type family RemoveColumn (name :: Symbol) (cols :: [(Symbol, Type)]) :: [(Symbol, Type)] where ...
- type family Impute (name :: Symbol) (cols :: [(Symbol, Type)]) :: [(Symbol, Type)] where ...
- type family SetColumnType (name :: Symbol) b (cols :: [(Symbol, Type)]) :: [(Symbol, Type)] where ...
- type family Append (xs :: [k]) (ys :: [k]) :: [k] where ...
- type family Reverse (xs :: [(Symbol, Type)]) :: [(Symbol, Type)] where ...
- type family StripAllMaybe (cols :: [(Symbol, Type)]) :: [(Symbol, Type)] where ...
- type family StripMaybeAt (name :: Symbol) (cols :: [(Symbol, Type)]) :: [(Symbol, Type)] where ...
- type family GroupKeyColumns (keys :: [Symbol]) (cols :: [(Symbol, Type)]) :: [(Symbol, Type)] where ...
- type family InnerJoinSchema (keys :: [Symbol]) (left :: [(Symbol, Type)]) (right :: [(Symbol, Type)]) :: [(Symbol, Type)] where ...
- type family LeftJoinSchema (keys :: [Symbol]) (left :: [(Symbol, Type)]) (right :: [(Symbol, Type)]) :: [(Symbol, Type)] where ...
- type family RightJoinSchema (keys :: [Symbol]) (left :: [(Symbol, Type)]) (right :: [(Symbol, Type)]) :: [(Symbol, Type)] where ...
- type family FullOuterJoinSchema (keys :: [Symbol]) (left :: [(Symbol, Type)]) (right :: [(Symbol, Type)]) :: [(Symbol, Type)] where ...
- type family AssertAbsent (name :: Symbol) (cols :: [(Symbol, Type)]) where ...
- type family AssertAllPresent (name :: [Symbol]) (cols :: [(Symbol, Type)]) where ...
- type family AssertPresent (name :: Symbol) (cols :: [(Symbol, Type)]) where ...
- type family AssertDisjoint (left :: [(Symbol, Type)]) (right :: [(Symbol, Type)]) where ...
- type family AssertRealColumn (fn :: Symbol) (name :: Symbol) a where ...
- type family AllColumnsReal (fn :: Symbol) (cols :: [(Symbol, Type)]) where ...
- type family IsRealType a :: Bool where ...
- class KnownSchema (cols :: [(Symbol, Type)]) where
- schemaEvidence :: [(Text, SomeTypeRep)]
- schemaColumnNames :: forall (cols :: [(Symbol, Type)]). KnownSchema cols => [Text]
- class AllKnownSymbol (names :: [Symbol]) where
- symbolVals :: [Text]
Core types
data TypedDataFrame (cols :: [(Symbol, Type)]) #
A phantom-typed wrapper over the untyped DataFrame.
The type parameter cols is a type-level list of '(name, ty) pairs
that tracks the schema at compile time. All operations delegate to the
untyped core at runtime and update the phantom type at compile time.
Instances
| Show (TypedDataFrame cols) | |
Defined in DataFrame.Typed.Types Methods showsPrec :: Int -> TypedDataFrame cols -> ShowS # show :: TypedDataFrame cols -> String # showList :: [TypedDataFrame cols] -> ShowS # | |
| ToDataFrame (TypedDataFrame cols) | |
Defined in DataFrame.Typed.Freeze Methods toDataFrame :: TypedDataFrame cols -> DataFrame # | |
| Eq (TypedDataFrame cols) | |
Defined in DataFrame.Typed.Types Methods (==) :: TypedDataFrame cols -> TypedDataFrame cols -> Bool # (/=) :: TypedDataFrame cols -> TypedDataFrame cols -> Bool # | |
data TypedGrouped (keys :: [Symbol]) (cols :: [(Symbol, Type)]) #
A phantom-typed wrapper over GroupedDataFrame.
Inline replacement for Data.These.These to keep dataframe-core free
of the these package dependency. Only the three constructors and the
derived classes are used internally.
Instances
| Foldable (These a) | |
Defined in DataFrame.Internal.Types Methods fold :: Monoid m => These a m -> m # foldMap :: Monoid m => (a0 -> m) -> These a a0 -> m # foldMap' :: Monoid m => (a0 -> m) -> These a a0 -> m # foldr :: (a0 -> b -> b) -> b -> These a a0 -> b # foldr' :: (a0 -> b -> b) -> b -> These a a0 -> b # foldl :: (b -> a0 -> b) -> b -> These a a0 -> b # foldl' :: (b -> a0 -> b) -> b -> These a a0 -> b # foldr1 :: (a0 -> a0 -> a0) -> These a a0 -> a0 # foldl1 :: (a0 -> a0 -> a0) -> These a a0 -> a0 # toList :: These a a0 -> [a0] # elem :: Eq a0 => a0 -> These a a0 -> Bool # maximum :: Ord a0 => These a a0 -> a0 # minimum :: Ord a0 => These a a0 -> a0 # | |
| Traversable (These a) | |
Defined in DataFrame.Internal.Types | |
| Functor (These a) | |
| (Read a, Read b) => Read (These a b) | |
| (Show a, Show b) => Show (These a b) | |
| (Eq a, Eq b) => Eq (These a b) | |
| (Ord a, Ord b) => Ord (These a b) | |
Typed expressions
newtype TExpr (cols :: [(Symbol, Type)]) a #
A typed expression validated against schema cols, producing values of type a.
Unlike the untyped 'Expr a', a TExpr can only be constructed through
type-safe combinators (col, lit, arithmetic operations) that verify
column references exist in the schema with the correct type.
Use unTExpr to extract the underlying Expr for delegation to the untyped API.
Instances
| Fit cfg [Expr Double] => Fit cfg [TExpr cols Double] | The same lift for the unsupervised feature-list inputs. | ||||||||
| Fit cfg (Expr a) => Fit cfg (TExpr cols a) | Lift any model fittable on an untyped target | ||||||||
Defined in DataFrame.Model Associated Types
| |||||||||
| Show a => Show (TExpr cols a) | Shows the underlying expression; the schema phantom is type-level only. | ||||||||
| type FrameReq cfg [TExpr cols Double] | |||||||||
| type ModelOf cfg [TExpr cols Double] | |||||||||
| type FrameReq cfg (TExpr cols a) | |||||||||
Defined in DataFrame.Model | |||||||||
| type ModelOf cfg (TExpr cols a) | |||||||||
Defined in DataFrame.Model | |||||||||
col :: forall (name :: Symbol) (cols :: [(Symbol, Type)]) a. (KnownSymbol name, a ~ SafeLookup name cols, Columnable a, AssertPresent name cols) => TExpr cols a #
Create a typed column reference. This is the key type-safety entry point.
The column name must exist in cols and its type must match a.
Both checks happen at compile time via type families.
salary :: TExpr '[("salary", Double)] Double
salary = col @"salary"
lit :: forall a (cols :: [(Symbol, Type)]). Columnable a => a -> TExpr cols a #
Create a literal expression. Valid for any schema since it references no columns.
ifThenElse :: forall a (cols :: [(Symbol, Type)]). Columnable a => TExpr cols Bool -> TExpr cols a -> TExpr cols a -> TExpr cols a #
Conditional expression.
lift :: forall a b (cols :: [(Symbol, Type)]). (Columnable a, Columnable b) => (a -> b) -> TExpr cols a -> TExpr cols b #
Lift a unary function into a typed expression.
lift2 :: forall a b c (cols :: [(Symbol, Type)]). (Columnable a, Columnable b, Columnable c) => (a -> b -> c) -> TExpr cols a -> TExpr cols b -> TExpr cols c #
Lift a binary function into typed expressions.
nullLift :: forall a r (cols :: [(Symbol, Type)]). (NullLift1Op a r (NullLift1Result a r), Columnable (NullLift1Result a r)) => (BaseType a -> r) -> TExpr cols a -> TExpr cols (NullLift1Result a r) #
Typed nullLift: lift a unary function with nullable propagation.
When the input is Maybe a, Nothing short-circuits; when plain a, applies directly.
The return type is inferred via NullLift1Result: no annotation needed.
nullLift2 :: forall a b r (cols :: [(Symbol, Type)]). (NullLift2Op a b r (NullLift2Result a b r), Columnable (NullLift2Result a b r)) => (BaseType a -> BaseType b -> r) -> TExpr cols a -> TExpr cols b -> TExpr cols (NullLift2Result a b r) #
Typed nullLift2: lift a binary function with nullable propagation.
Any Nothing operand short-circuits to Nothing in the result.
The return type is inferred via NullLift2Result: no annotation needed.
Same-type comparison operators
(.==.) :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Eq a) => TExpr cols a -> TExpr cols a -> TExpr cols Bool infixl 4 #
(./=.) :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Eq a) => TExpr cols a -> TExpr cols a -> TExpr cols Bool infixl 4 #
(.<.) :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TExpr cols a -> TExpr cols Bool infixl 4 #
(.<=.) :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TExpr cols a -> TExpr cols Bool infixl 4 #
(.>=.) :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TExpr cols a -> TExpr cols Bool infixl 4 #
(.>.) :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TExpr cols a -> TExpr cols Bool infixl 4 #
Nullable-aware arithmetic operators
(.+) :: forall a b (cols :: [(Symbol, Type)]). (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b (Promote (BaseType a) (BaseType b)) (WidenResult a b), Num (Promote (BaseType a) (BaseType b))) => TExpr cols a -> TExpr cols b -> TExpr cols (WidenResult a b) infixl 6 #
Nullable-aware addition. Works for all combinations of nullable/non-nullable operands.
col @"x" .+ col @"y" -- :: TExpr cols (Maybe Int) when y :: Maybe Int
(.-) :: forall a b (cols :: [(Symbol, Type)]). (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b (Promote (BaseType a) (BaseType b)) (WidenResult a b), Num (Promote (BaseType a) (BaseType b))) => TExpr cols a -> TExpr cols b -> TExpr cols (WidenResult a b) infixl 6 #
Nullable-aware subtraction.
(.*) :: forall a b (cols :: [(Symbol, Type)]). (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b (Promote (BaseType a) (BaseType b)) (WidenResult a b), Num (Promote (BaseType a) (BaseType b))) => TExpr cols a -> TExpr cols b -> TExpr cols (WidenResult a b) infixl 7 #
Nullable-aware multiplication.
(./) :: forall a b (cols :: [(Symbol, Type)]). (DivWidenOp (BaseType a) (BaseType b), NullLift2Op a b (PromoteDiv (BaseType a) (BaseType b)) (WidenResultDiv a b), Fractional (PromoteDiv (BaseType a) (BaseType b))) => TExpr cols a -> TExpr cols b -> TExpr cols (WidenResultDiv a b) infixl 7 #
Nullable-aware division. Integral operands are promoted to Double.
Nullable-aware comparison operators (three-valued logic)
(.==) :: forall a b (cols :: [(Symbol, Type)]). (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Eq (Promote (BaseType a) (BaseType b))) => TExpr cols a -> TExpr cols b -> TExpr cols (NullCmpResult a b) infix 4 #
Nullable-aware equality. Widens numeric operands to their common type,
so TExpr cols Double .== TExpr cols Int typechecks. Returns Maybe Bool
when either operand is nullable.
(./=) :: forall a b (cols :: [(Symbol, Type)]). (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Eq (Promote (BaseType a) (BaseType b))) => TExpr cols a -> TExpr cols b -> TExpr cols (NullCmpResult a b) infix 4 #
Nullable-aware inequality. Widens numeric operands to their common type.
(.<) :: forall a b (cols :: [(Symbol, Type)]). (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Ord (Promote (BaseType a) (BaseType b))) => TExpr cols a -> TExpr cols b -> TExpr cols (NullCmpResult a b) infix 4 #
Nullable-aware less-than. Widens numeric operands to their common type.
(.<=) :: forall a b (cols :: [(Symbol, Type)]). (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Ord (Promote (BaseType a) (BaseType b))) => TExpr cols a -> TExpr cols b -> TExpr cols (NullCmpResult a b) infix 4 #
Nullable-aware less-than-or-equal. Widens numeric operands to their common type.
(.>=) :: forall a b (cols :: [(Symbol, Type)]). (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Ord (Promote (BaseType a) (BaseType b))) => TExpr cols a -> TExpr cols b -> TExpr cols (NullCmpResult a b) infix 4 #
Nullable-aware greater-than-or-equal. Widens numeric operands to their common type.
(.>) :: forall a b (cols :: [(Symbol, Type)]). (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Ord (Promote (BaseType a) (BaseType b))) => TExpr cols a -> TExpr cols b -> TExpr cols (NullCmpResult a b) infix 4 #
Nullable-aware greater-than. Widens numeric operands to their common type.
Logical operators
(.&&.) :: forall (cols :: [(Symbol, Type)]). TExpr cols Bool -> TExpr cols Bool -> TExpr cols Bool infixr 3 #
(.||.) :: forall (cols :: [(Symbol, Type)]). TExpr cols Bool -> TExpr cols Bool -> TExpr cols Bool infixr 2 #
Aggregation expression combinators
mean :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a) => TExpr cols a -> TExpr cols Double #
median :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a, Unbox a) => TExpr cols a -> TExpr cols Double #
countAll :: forall (cols :: [(Symbol, Type)]). TExpr cols Int #
Row count, the equivalent of SQL's COUNT(*).
minimum :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TExpr cols a #
maximum :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TExpr cols a #
over :: forall (names :: [Symbol]) (cols :: [(Symbol, Type)]) a. (Columnable a, AllKnownSymbol names, AssertAllPresent names cols) => TExpr cols a -> TExpr cols a #
Expression combinators (full DataFrame.Functions parity)
div :: forall a (cols :: [(Symbol, Type)]). (Integral a, Columnable a) => TExpr cols a -> TExpr cols a -> TExpr cols a #
Integer division.
mod :: forall a (cols :: [(Symbol, Type)]). (Integral a, Columnable a) => TExpr cols a -> TExpr cols a -> TExpr cols a #
Integer modulus.
mode :: forall a (cols :: [(Symbol, Type)]). (Ord a, Columnable a, Eq a) => TExpr cols a -> TExpr cols a #
Most frequent value (aggregation).
sumMaybe :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Num a) => TExpr cols (Maybe a) -> TExpr cols a #
Sum of a nullable column, ignoring Nothing (aggregation).
meanMaybe :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a) => TExpr cols (Maybe a) -> TExpr cols Double #
Mean of a nullable column, ignoring Nothing (aggregation).
variance :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a, Unbox a) => TExpr cols a -> TExpr cols Double #
Variance (aggregation).
medianMaybe :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a) => TExpr cols (Maybe a) -> TExpr cols Double #
Median of a nullable column, ignoring Nothing (aggregation).
percentile :: forall (cols :: [(Symbol, Type)]). Int -> TExpr cols Double -> TExpr cols Double #
The n-th percentile (aggregation).
stddev :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a, Unbox a) => TExpr cols a -> TExpr cols Double #
Standard deviation (aggregation).
stddevMaybe :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a) => TExpr cols (Maybe a) -> TExpr cols Double #
Standard deviation of a nullable column, ignoring Nothing (aggregation).
zScore :: forall (cols :: [(Symbol, Type)]). TExpr cols Double -> TExpr cols Double #
Z-score (value minus group mean, over standard deviation).
pow :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Num a) => TExpr cols a -> Int -> TExpr cols a #
Raise an expression to an integer power.
relu :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Num a, Ord a) => TExpr cols a -> TExpr cols a #
Rectified linear unit: max 0.
min :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TExpr cols a -> TExpr cols a #
Element-wise minimum of two expressions.
max :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TExpr cols a -> TExpr cols a #
Element-wise maximum of two expressions.
reduce :: forall a b (cols :: [(Symbol, Type)]). (Columnable a, Columnable b) => TExpr cols b -> a -> (a -> b -> a) -> TExpr cols a #
Fold a column into a single value with a seed and step function (aggregation).
toMaybe :: forall a (cols :: [(Symbol, Type)]). Columnable a => TExpr cols a -> TExpr cols (Maybe a) #
Wrap each value in Just.
fromMaybe :: forall a (cols :: [(Symbol, Type)]). Columnable a => a -> TExpr cols (Maybe a) -> TExpr cols a #
Replace Nothing with a default.
isJust :: forall a (cols :: [(Symbol, Type)]). Columnable a => TExpr cols (Maybe a) -> TExpr cols Bool #
True where the value is Just.
isNothing :: forall a (cols :: [(Symbol, Type)]). Columnable a => TExpr cols (Maybe a) -> TExpr cols Bool #
True where the value is Nothing.
fromJust :: forall a (cols :: [(Symbol, Type)]). Columnable a => TExpr cols (Maybe a) -> TExpr cols a #
whenPresent :: forall a b (cols :: [(Symbol, Type)]). (Columnable a, Columnable b) => (a -> b) -> TExpr cols (Maybe a) -> TExpr cols (Maybe b) #
Apply a function only where the value is present.
whenBothPresent :: forall a b c (cols :: [(Symbol, Type)]). (Columnable a, Columnable b, Columnable c) => (a -> b -> c) -> TExpr cols (Maybe a) -> TExpr cols (Maybe b) -> TExpr cols (Maybe c) #
Apply a binary function only where both values are present.
recode :: forall a b (cols :: [(Symbol, Type)]). (Columnable a, Columnable b, Show a, Show b, Show (a, b)) => [(a, b)] -> TExpr cols a -> TExpr cols (Maybe b) #
Map values through a lookup table, yielding Nothing for misses.
recodeWithCondition :: forall a b (cols :: [(Symbol, Type)]). (Columnable a, Columnable b) => TExpr cols b -> [(TExpr cols a -> TExpr cols Bool, b)] -> TExpr cols a -> TExpr cols b #
Pick the first value whose condition holds, else a fallback.
recodeWithDefault :: forall a b (cols :: [(Symbol, Type)]). (Columnable a, Columnable b, Show (a, b)) => b -> [(a, b)] -> TExpr cols a -> TExpr cols b #
Map values through a lookup table, with a default for misses.
firstOrNothing :: forall a (cols :: [(Symbol, Type)]). Columnable a => TExpr cols [a] -> TExpr cols (Maybe a) #
First element of a list column, or Nothing.
lastOrNothing :: forall a (cols :: [(Symbol, Type)]). Columnable a => TExpr cols [a] -> TExpr cols (Maybe a) #
Last element of a list column, or Nothing.
splitOn :: forall (cols :: [(Symbol, Type)]). Text -> TExpr cols Text -> TExpr cols [Text] #
Split text on a delimiter.
match :: forall (cols :: [(Symbol, Type)]). Text -> TExpr cols Text -> TExpr cols (Maybe Text) #
First regex match, or Nothing.
matchAll :: forall (cols :: [(Symbol, Type)]). Text -> TExpr cols Text -> TExpr cols [Text] #
All regex matches.
parseDate :: forall t (cols :: [(Symbol, Type)]). (ParseTime t, Columnable t) => Text -> TExpr cols Text -> TExpr cols (Maybe t) #
Parse text into a time value with the given format.
daysBetween :: forall (cols :: [(Symbol, Type)]). TExpr cols Day -> TExpr cols Day -> TExpr cols Int #
Number of days between two dates.
bind :: forall a m b (cols :: [(Symbol, Type)]). (Columnable a, Columnable (m a), Monad m, Columnable b, Columnable (m b)) => (a -> m b) -> TExpr cols (m a) -> TExpr cols (m b) #
Monadic bind over a column of monadic values.
Cast / coercion expressions
castExpr :: forall b (cols :: [(Symbol, Type)]) src. (Columnable b, Columnable src, Read b) => TExpr cols src -> TExpr cols (Maybe b) #
castExprWithDefault :: forall b (cols :: [(Symbol, Type)]) src. (Columnable b, Columnable src, Read b) => b -> TExpr cols src -> TExpr cols b #
castExprEither :: forall b (cols :: [(Symbol, Type)]) src. (Columnable b, Columnable src, Read b) => TExpr cols src -> TExpr cols (Either Text b) #
unsafeCastExpr :: forall b (cols :: [(Symbol, Type)]) src. (Columnable b, Columnable src, Read b) => TExpr cols src -> TExpr cols b #
toDouble :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a) => TExpr cols a -> TExpr cols Double #
Typed sort orders
data TSortOrder (cols :: [(Symbol, Type)]) where #
A typed sort order validated against schema cols.
Constructors
| Asc :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TSortOrder cols | |
| Desc :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TSortOrder cols |
asc :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TSortOrder cols #
Create an ascending sort order from a typed expression.
desc :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TSortOrder cols #
Create a descending sort order from a typed expression.
Freeze / thaw boundary
freeze :: forall (cols :: [(Symbol, Type)]). KnownSchema cols => DataFrame -> Maybe (TypedDataFrame cols) #
Validate that an untyped DataFrame matches the expected schema cols,
then wrap it. Returns Nothing on mismatch.
freezeWithError :: forall (cols :: [(Symbol, Type)]). KnownSchema cols => DataFrame -> Either Text (TypedDataFrame cols) #
Like freeze but returns a descriptive error message on failure.
thaw :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> DataFrame #
Unwrap a typed DataFrame back to the untyped representation. Always safe; discards type information.
unsafeFreeze :: forall (cols :: [(Symbol, Type)]). DataFrame -> TypedDataFrame cols #
Wrap an untyped DataFrame without any validation. Used internally after delegation where the library guarantees schema correctness.
Typed column access
columnAsVector :: forall (name :: Symbol) (cols :: [(Symbol, Type)]) a. (KnownSymbol name, a ~ SafeLookup name cols, Columnable a, AssertPresent name cols) => TypedDataFrame cols -> Vector a #
Retrieve a column as a boxed Vector, with the type determined by
the schema. The column must exist (enforced at compile time).
columnAsList :: forall (name :: Symbol) (cols :: [(Symbol, Type)]) a. (KnownSymbol name, a ~ SafeLookup name cols, Columnable a, AssertPresent name cols) => TypedDataFrame cols -> [a] #
Retrieve a column as a list, with the type determined by the schema.
columnAsIntVector :: forall (name :: Symbol) (cols :: [(Symbol, Type)]) a. (KnownSymbol name, a ~ SafeLookup name cols, Columnable a, AssertRealColumn "columnAsIntVector" name a, Real a, Unbox a, AssertPresent name cols) => TypedDataFrame cols -> Vector Int #
Retrieve a column coerced to an unboxed Int vector, named by type
application. The column must exist and be numeric — both are compile-time
checks via SafeLookup, so this is total (no Either, no runtime throw).
columnAsDoubleVector :: forall (name :: Symbol) (cols :: [(Symbol, Type)]) a. (KnownSymbol name, a ~ SafeLookup name cols, Columnable a, AssertRealColumn "columnAsDoubleVector" name a, Real a, Unbox a, AssertPresent name cols) => TypedDataFrame cols -> Vector Double #
Retrieve a column coerced to an unboxed Double vector. See columnAsIntVector.
columnAsFloatVector :: forall (name :: Symbol) (cols :: [(Symbol, Type)]) a. (KnownSymbol name, a ~ SafeLookup name cols, Columnable a, AssertRealColumn "columnAsFloatVector" name a, Real a, Unbox a, AssertPresent name cols) => TypedDataFrame cols -> Vector Float #
Retrieve a column coerced to an unboxed Float vector. See columnAsIntVector.
columnAsUnboxedVector :: forall (name :: Symbol) (cols :: [(Symbol, Type)]) a. (KnownSymbol name, a ~ SafeLookup name cols, Columnable a, Unbox a, AssertPresent name cols) => TypedDataFrame cols -> Vector a #
Retrieve a column as an unboxed vector of its own element type. The column must exist and be unboxable — both compile-time checks, so this is total.
toDoubleMatrix :: forall (cols :: [(Symbol, Type)]). AllColumnsReal "toDoubleMatrix" cols => TypedDataFrame cols -> Vector (Vector Double) #
Convert every column to Double and transpose into a row-major matrix.
Total: AllColumnsReal proves at compile time that every column is numeric and
unboxed, so the conversion cannot fail.
toFloatMatrix :: forall (cols :: [(Symbol, Type)]). AllColumnsReal "toFloatMatrix" cols => TypedDataFrame cols -> Vector (Vector Float) #
Convert every column to Float and transpose into a row-major matrix. See toDoubleMatrix.
toIntMatrix :: forall (cols :: [(Symbol, Type)]). AllColumnsReal "toIntMatrix" cols => TypedDataFrame cols -> Vector (Vector Int) #
Convert every column to Int and transpose into a row-major matrix. See toDoubleMatrix.
Schema-preserving operations
filterWhere :: forall (cols :: [(Symbol, Type)]). TExpr cols Bool -> TypedDataFrame cols -> TypedDataFrame cols #
Filter rows where a boolean expression evaluates to True. The expression is validated against the schema at compile time.
filter :: forall a (cols :: [(Symbol, Type)]). Columnable a => TExpr cols a -> (a -> Bool) -> TypedDataFrame cols -> TypedDataFrame cols #
Filter rows by applying a predicate to a typed expression.
filterBy :: forall a (cols :: [(Symbol, Type)]). Columnable a => (a -> Bool) -> TExpr cols a -> TypedDataFrame cols -> TypedDataFrame cols #
Filter rows by a predicate on a column expression (flipped argument order).
filterAllJust :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> TypedDataFrame (StripAllMaybe cols) #
Keep only rows where ALL Optional columns have Just values.
Strips Maybe from all column types in the result schema.
df :: TDF '[ '("x", Maybe Double), '("y", Int)]
filterAllJust df :: TDF '[ '("x", Double), '("y", Int)]
filterJust :: forall (name :: Symbol) (cols :: [(Symbol, Type)]). (KnownSymbol name, AssertPresent name cols) => TypedDataFrame cols -> TypedDataFrame (StripMaybeAt name cols) #
Keep only rows where the named column has Just values.
Strips Maybe from that column's type in the result schema.
filterJust @"x" df
filterNothing :: forall (name :: Symbol) (cols :: [(Symbol, Type)]). (KnownSymbol name, AssertPresent name cols) => TypedDataFrame cols -> TypedDataFrame cols #
Keep only rows where the named column has Nothing. Schema is preserved (column types unchanged, just fewer rows).
filterAllNothing :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> TypedDataFrame cols #
Keep only rows where every nullable column has Nothing. Schema is preserved.
sortBy :: forall (cols :: [(Symbol, Type)]). [TSortOrder cols] -> TypedDataFrame cols -> TypedDataFrame cols #
Sort by the given typed sort orders. Sort orders reference columns that are validated against the schema.
take :: forall (cols :: [(Symbol, Type)]). Int -> TypedDataFrame cols -> TypedDataFrame cols #
Take the first n rows.
takeLast :: forall (cols :: [(Symbol, Type)]). Int -> TypedDataFrame cols -> TypedDataFrame cols #
Take the last n rows.
drop :: forall (cols :: [(Symbol, Type)]). Int -> TypedDataFrame cols -> TypedDataFrame cols #
Drop the first n rows.
dropLast :: forall (cols :: [(Symbol, Type)]). Int -> TypedDataFrame cols -> TypedDataFrame cols #
Drop the last n rows.
range :: forall (cols :: [(Symbol, Type)]). (Int, Int) -> TypedDataFrame cols -> TypedDataFrame cols #
Take rows in the given range (start, end).
cube :: forall (cols :: [(Symbol, Type)]). (Int, Int) -> TypedDataFrame cols -> TypedDataFrame cols #
Take a sub-cube of the DataFrame.
distinct :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> TypedDataFrame cols #
Remove duplicate rows.
sample :: forall g (cols :: [(Symbol, Type)]). RandomGen g => g -> Double -> TypedDataFrame cols -> TypedDataFrame cols #
Randomly sample a fraction of rows.
shuffle :: forall g (cols :: [(Symbol, Type)]). RandomGen g => g -> TypedDataFrame cols -> TypedDataFrame cols #
Shuffle all rows randomly.
Schema-modifying operations
derive :: forall (name :: Symbol) a (cols :: [(Symbol, Type)]). (KnownSymbol name, Columnable a, AssertAbsent name cols) => TExpr cols a -> TypedDataFrame cols -> TypedDataFrame (Snoc cols '(name, a)) #
Derive a new column from a typed expression. The column name must NOT
already exist in the schema (enforced at compile time via AssertAbsent).
The expression is validated against the current schema.
df' = derive @"total" (col @"price" * col @"qty") df
-- df' :: TDF ('("total", Double ': originalCols))
impute :: forall (name :: Symbol) a (cols :: [(Symbol, Type)]). (KnownSymbol name, Columnable a, Maybe a ~ Lookup name cols) => a -> TypedDataFrame cols -> TypedDataFrame (Impute name cols) #
select :: forall (names :: [Symbol]) (cols :: [(Symbol, Type)]). (AllKnownSymbol names, AssertAllPresent names cols) => TypedDataFrame cols -> TypedDataFrame (SubsetSchema names cols) #
Select a subset of columns by name.
exclude :: forall (names :: [Symbol]) (cols :: [(Symbol, Type)]). AllKnownSymbol names => TypedDataFrame cols -> TypedDataFrame (ExcludeSchema names cols) #
Exclude columns by name.
rename :: forall (old :: Symbol) (new :: Symbol) (cols :: [(Symbol, Type)]). (KnownSymbol old, KnownSymbol new) => TypedDataFrame cols -> TypedDataFrame (RenameInSchema old new cols) #
Rename a column.
renameMany :: forall (pairs :: [(Symbol, Symbol)]) (cols :: [(Symbol, Type)]). AllKnownPairs pairs => TypedDataFrame cols -> TypedDataFrame (RenameManyInSchema pairs cols) #
Rename multiple columns from a type-level list of pairs.
insert :: forall (name :: Symbol) a (cols :: [(Symbol, Type)]) t. (KnownSymbol name, Columnable a, Foldable t, AssertAbsent name cols) => t a -> TypedDataFrame cols -> TypedDataFrame ('(name, a) ': cols) #
Insert a new column from a Foldable container.
insertColumn :: forall (name :: Symbol) a (cols :: [(Symbol, Type)]). (KnownSymbol name, Columnable a, AssertAbsent name cols) => Column -> TypedDataFrame cols -> TypedDataFrame ('(name, a) ': cols) #
Insert a raw Column value.
insertVector :: forall (name :: Symbol) a (cols :: [(Symbol, Type)]). (KnownSymbol name, Columnable a, AssertAbsent name cols) => Vector a -> TypedDataFrame cols -> TypedDataFrame ('(name, a) ': cols) #
Insert a boxed Vector.
cloneColumn :: forall (old :: Symbol) (new :: Symbol) (cols :: [(Symbol, Type)]). (KnownSymbol old, KnownSymbol new, AssertPresent old cols, AssertAbsent new cols) => TypedDataFrame cols -> TypedDataFrame ('(new, Lookup old cols) ': cols) #
Clone an existing column under a new name.
dropColumn :: forall (name :: Symbol) (cols :: [(Symbol, Type)]). (KnownSymbol name, AssertPresent name cols) => TypedDataFrame cols -> TypedDataFrame (RemoveColumn name cols) #
Drop a column by name.
replaceColumn :: forall (name :: Symbol) a (cols :: [(Symbol, Type)]). (KnownSymbol name, Columnable a, a ~ SafeLookup name cols, AssertPresent name cols) => TExpr cols a -> TypedDataFrame cols -> TypedDataFrame cols #
Replace an existing column with new values derived from a typed expression. The column must already exist and the new type must match.
Metadata
dimensions :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> (Int, Int) #
columnNames :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> [Text] #
Vertical merge
append :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> TypedDataFrame cols -> TypedDataFrame cols #
Vertically merge two DataFrames with the same schema.
Set algebra (topos operations)
union :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> TypedDataFrame cols -> TypedDataFrame cols #
Rows appearing in either DataFrame, deduplicated (set union).
intersect :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> TypedDataFrame cols -> TypedDataFrame cols #
Rows appearing in both DataFrames, deduplicated (set intersection).
difference :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> TypedDataFrame cols -> TypedDataFrame cols #
Rows in the left DataFrame but not the right, deduplicated
(relational EXCEPT; the subobject complement).
symmetricDifference :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> TypedDataFrame cols -> TypedDataFrame cols #
Rows in exactly one of the two DataFrames, deduplicated.
Joins
innerJoin :: forall (keys :: [Symbol]) (left :: [(Symbol, Type)]) (right :: [(Symbol, Type)]). (AllKnownSymbol keys, AssertAllPresent keys left, AssertAllPresent keys right, AssertKeyTypesMatch keys left right) => TypedDataFrame left -> TypedDataFrame right -> TypedDataFrame (InnerJoinSchema keys left right) #
Typed inner join on one or more key columns.
leftJoin :: forall (keys :: [Symbol]) (left :: [(Symbol, Type)]) (right :: [(Symbol, Type)]). (AllKnownSymbol keys, AssertAllPresent keys left, AssertAllPresent keys right, AssertKeyTypesMatch keys left right) => TypedDataFrame left -> TypedDataFrame right -> TypedDataFrame (LeftJoinSchema keys left right) #
Typed left join.
rightJoin :: forall (keys :: [Symbol]) (left :: [(Symbol, Type)]) (right :: [(Symbol, Type)]). (AllKnownSymbol keys, AssertAllPresent keys left, AssertAllPresent keys right, AssertKeyTypesMatch keys left right) => TypedDataFrame left -> TypedDataFrame right -> TypedDataFrame (RightJoinSchema keys left right) #
Typed right join.
fullOuterJoin :: forall (keys :: [Symbol]) (left :: [(Symbol, Type)]) (right :: [(Symbol, Type)]). (AllKnownSymbol keys, AssertAllPresent keys left, AssertAllPresent keys right, AssertKeyTypesMatch keys left right) => TypedDataFrame left -> TypedDataFrame right -> TypedDataFrame (FullOuterJoinSchema keys left right) #
Typed full outer join.
GroupBy and Aggregation
groupBy :: forall (keys :: [Symbol]) (cols :: [(Symbol, Type)]). (AllKnownSymbol keys, AssertAllPresent keys cols) => TypedDataFrame cols -> TypedGrouped keys cols #
Group a typed DataFrame by one or more key columns.
grouped = groupBy @'["department"] employees
as :: forall (name :: Symbol) a (keys :: [Symbol]) (cols :: [(Symbol, Type)]) (aggs :: [(Symbol, Type)]). (KnownSymbol name, Columnable a) => TExpr cols a -> TAgg keys cols aggs -> TAgg keys cols ('(name, a) ': aggs) #
Build a named aggregation entry. The result column name is supplied via
TypeApplications; the underlying expression is validated against the
source schema at compile time.
as produces a transformer on the aggregation chain — entries compose
with plain (.) from Prelude (or via (|>) for SQL-like postfix
reading). aggregate applies the composed transformer to the empty chain
internally, so no terminator is needed.
Prefix form
result = grouped |> aggregate
( as @"total" (sum (col @"amount"))
. as @"orders" (count (col @"order_id"))
. as @"avg" (mean (col @"amount"))
)
Postfix form (SQL-like)
result = grouped |> aggregate
( (sum (col @"amount") |> as @"total")
. (count (col @"order_id") |> as @"orders")
. (mean (col @"amount") |> as @"avg")
)
Per-entry parentheses are required in the postfix form because
(.) binds tighter than (|>).
aggregate :: forall (keys :: [Symbol]) (cols :: [(Symbol, Type)]) (aggs :: [(Symbol, Type)]). (TAgg keys cols ('[] :: [(Symbol, Type)]) -> TAgg keys cols aggs) -> TypedGrouped keys cols -> TypedDataFrame (Append (GroupKeyColumns keys cols) (Reverse aggs)) #
Run a typed aggregation against a grouped DataFrame.
The first argument is a chain of as entries composed with (.). The
empty composition (id) yields just the group keys. The result schema is
the group-key columns followed by the aggregation columns in declaration
order.
result = grouped |> aggregate
( as @"total" (sum (col @"amount"))
. as @"orders" (count (col @"order_id"))
)
-- result :: TypedDataFrame
-- '[ '("region", Text)
-- , '("total", Double)
-- , '("orders", Int)
-- ]
aggregateUntyped :: forall (keys :: [Symbol]) (cols :: [(Symbol, Type)]). [NamedExpr] -> TypedGrouped keys cols -> DataFrame #
Escape hatch: run an untyped aggregation and return a raw DataFrame.
Column transformations
applyColumn :: forall (name :: Symbol) a b (cols :: [(Symbol, Type)]). (KnownSymbol name, a ~ SafeLookup name cols, Columnable a, Columnable b, AssertPresent name cols) => (a -> b) -> TypedDataFrame cols -> TypedDataFrame (SetColumnType name b cols) #
Map a function over a column, rewriting its element type from a to b.
The schema's entry for name is updated via SetColumnType.
df' = applyColumn @"age" (show :: Int -> String) df -- the "age" column is now String-typed
applyMany :: forall (names :: [Symbol]) a (cols :: [(Symbol, Type)]). (AllKnownSymbol names, Columnable a, AssertAllColumnsHaveType names a cols) => (a -> a) -> TypedDataFrame cols -> TypedDataFrame cols #
Apply a type-preserving function to several columns at once. Every named
column must already share the element type a (enforced by
AssertAllColumnsHaveType).
applyWhere :: forall (filterName :: Symbol) (targetName :: Symbol) a b (cols :: [(Symbol, Type)]). (KnownSymbol filterName, KnownSymbol targetName, a ~ SafeLookup filterName cols, b ~ SafeLookup targetName cols, Columnable a, Columnable b, AssertPresent filterName cols, AssertPresent targetName cols) => (a -> Bool) -> (b -> b) -> TypedDataFrame cols -> TypedDataFrame cols #
Apply a function to a target column only on rows where a condition holds on a filter column. Both columns are named by type application; the target keeps its type.
applyWhere @"flagged" @"score" id (* 2) df
applyAtIndex :: forall (name :: Symbol) a (cols :: [(Symbol, Type)]). (KnownSymbol name, a ~ SafeLookup name cols, Columnable a, AssertPresent name cols) => Int -> (a -> a) -> TypedDataFrame cols -> TypedDataFrame cols #
Apply a type-preserving function to a single row of a column.
safeApply :: forall (name :: Symbol) a b (cols :: [(Symbol, Type)]). (KnownSymbol name, a ~ SafeLookup name cols, Columnable a, Columnable b, AssertPresent name cols) => (a -> b) -> TypedDataFrame cols -> Either DataFrameException (TypedDataFrame (SetColumnType name b cols)) #
Like applyColumn but returns the error instead of throwing.
deriveWithExpr :: forall (name :: Symbol) a (cols :: [(Symbol, Type)]). (KnownSymbol name, Columnable a, AssertAbsent name cols) => TExpr cols a -> TypedDataFrame cols -> (TExpr (Snoc cols '(name, a)) a, TypedDataFrame (Snoc cols '(name, a))) #
Derive a new column and also return a typed reference to it. The returned expression lives in the extended schema, so it can feed later operations.
insertWithDefault :: forall (name :: Symbol) a (cols :: [(Symbol, Type)]) t. (KnownSymbol name, Columnable a, Foldable t, AssertAbsent name cols) => a -> t a -> TypedDataFrame cols -> TypedDataFrame ('(name, a) ': cols) #
Insert a column from a Foldable, padding missing rows with a default.
insertVectorWithDefault :: forall (name :: Symbol) a (cols :: [(Symbol, Type)]). (KnownSymbol name, Columnable a, AssertAbsent name cols) => a -> Vector a -> TypedDataFrame cols -> TypedDataFrame ('(name, a) ': cols) #
Insert a boxed Vector, padding missing rows with a default.
insertUnboxedVector :: forall (name :: Symbol) a (cols :: [(Symbol, Type)]). (KnownSymbol name, Columnable a, Unbox a, AssertAbsent name cols) => Vector a -> TypedDataFrame cols -> TypedDataFrame ('(name, a) ': cols) #
Insert an unboxed Vector as a new column.
(|||) :: forall (left :: [(Symbol, Type)]) (right :: [(Symbol, Type)]). AssertDisjoint left right => TypedDataFrame left -> TypedDataFrame right -> TypedDataFrame (Append left right) #
Horizontal merge: place two DataFrames side by side. The schemas must be
disjoint (no shared column names), enforced by AssertDisjoint; the result
schema is their concatenation.
Sampling and splitting
randomSplit :: forall g (cols :: [(Symbol, Type)]). RandomGen g => g -> Double -> TypedDataFrame cols -> (TypedDataFrame cols, TypedDataFrame cols) #
Split rows into two DataFrames by a fraction.
kFolds :: forall g (cols :: [(Symbol, Type)]). RandomGen g => g -> Int -> TypedDataFrame cols -> [TypedDataFrame cols] #
Partition rows into k folds.
selectRows :: forall (cols :: [(Symbol, Type)]). [Int] -> TypedDataFrame cols -> TypedDataFrame cols #
Select rows by index.
| This may fail if the indices are out of bounds;
| use with caution or use filter to select rows by a predicate instead.
stratifiedSample :: forall g a (cols :: [(Symbol, Type)]). (SplittableGen g, Columnable a) => g -> Double -> TExpr cols a -> TypedDataFrame cols -> TypedDataFrame cols #
Sample a fraction of rows, preserving the distribution of a strata column.
stratifiedSplit :: forall g a (cols :: [(Symbol, Type)]). (SplittableGen g, Columnable a) => g -> Double -> TExpr cols a -> TypedDataFrame cols -> (TypedDataFrame cols, TypedDataFrame cols) #
Split rows by a fraction, preserving the distribution of a strata column.
Frequencies
valueCounts :: forall a (cols :: [(Symbol, Type)]). (Ord a, Columnable a) => TExpr cols a -> TypedDataFrame cols -> [(a, Int)] #
Count occurrences of each distinct value in a column.
valueProportions :: forall a (cols :: [(Symbol, Type)]). (Ord a, Columnable a) => TExpr cols a -> TypedDataFrame cols -> [(a, Double)] #
Proportion of each distinct value in a column.
Template Haskell
deriveSchema :: String -> DataFrame -> DecsQ #
deriveSchemaFromCsvFile :: String -> String -> DecsQ #
deriveSchemaFromCsvFileWith :: ReadOptions -> String -> String -> DecsQ #
deriveSchemaFromParquetFile :: String -> String -> DecsQ #
Derive a typed schema synonym from a Parquet file (or directory/glob).
deriveSchemaFromType :: Name -> DecsQ #
deriveSchemaFromTypeWith :: SchemaOptions -> Name -> DecsQ #
data SchemaOptions #
Options controlling deriveSchemaFromTypeWith.
Constructors
| SchemaOptions | |
Fields
| |
Record bridge (ADT - TypedDataFrame)
Bridge a Haskell record type a to a typed-dataframe schema.
The schema is exposed as an associated type family Schema so that
instances can pick it up from a Rep computation (see
SchemaOf) or from an explicit list emitted by
deriveSchemaFromType.
toColumns explodes a list of records into a list of named columns.
fromColumns reconstructs the records from a DataFrame, returning
Left err if a column is missing or has the wrong type.
fromRecordsTyped :: HasSchema a => [a] -> TypedDataFrame (Schema a) #
Like fromRecords but returns a TypedDataFrame tagged with the schema.
toRecordsTyped :: HasSchema a => TypedDataFrame (Schema a) -> Either Text [a] #
Like toRecords but accepts a TypedDataFrame.
Generics opt-in for schema derivation
type SchemaOf a = RepToSchema 'SnakeCase (Rep a) #
Snake_case schema derived from a's Generic representation.
type SchemaOfRaw a = RepToSchema 'IdentityCase (Rep a) #
Identity-cased schema derived from a's Generic representation.
Field-name policy applied to record selectors when computing
RepToSchema.
SnakeCase— translatecamelCaseFieldto"camel_case_field".IdentityCase— keep the selector name verbatim.
Constructors
| SnakeCase | |
| IdentityCase |
genericToColumns :: (Generic a, GHasColumns (Rep a)) => [a] -> [(Text, Column)] #
Default implementation of toColumns for any
Generic record. Field names are translated with camelCase -> snake_case.
instance HasSchema Order (SchemaOf Order) where toColumns = genericToColumns fromColumns = genericFromColumns
genericFromColumns :: (Generic a, GHasColumns (Rep a)) => DataFrame -> Either Text [a] #
Default implementation of fromColumns for any
Generic record.
Schema type families (for advanced use)
type family Lookup (name :: Symbol) (cols :: [(Symbol, Type)]) where ... #
Look up the element type of a column by name.
type family SafeLookup (name :: Symbol) (cols :: [(Symbol, Type)]) where ... #
Like Lookup, but returns a harmless fallback (Int) instead of
TypeError when the column is not found. Use together with
AssertPresent so the error fires exactly once.
Equations
| SafeLookup name ('(name, a) ': _1) = a | |
| SafeLookup name (_1 ': rest) = SafeLookup name rest | |
| SafeLookup name ('[] :: [(Symbol, Type)]) = Int |
type family HasName (name :: Symbol) (cols :: [(Symbol, Type)]) :: Bool where ... #
Check whether a column name exists in a schema (type-level Bool).
type family SubsetSchema (names :: [Symbol]) (cols :: [(Symbol, Type)]) :: [(Symbol, Type)] where ... #
Select a subset of columns by a list of names.
Equations
| SubsetSchema ('[] :: [Symbol]) cols = '[] :: [(Symbol, Type)] | |
| SubsetSchema (n ': ns) cols = '(n, Lookup n cols) ': SubsetSchema ns cols |
type family ExcludeSchema (names :: [Symbol]) (cols :: [(Symbol, Type)]) :: [(Symbol, Type)] where ... #
Exclude columns by a list of names.
Equations
| ExcludeSchema names ('[] :: [(Symbol, Type)]) = '[] :: [(Symbol, Type)] | |
| ExcludeSchema names ('(n, a) ': rest) = ExcludeSchemaHelper (IsElem n names) n a names rest |
type family RenameInSchema (old :: Symbol) (new :: Symbol) (cols :: [(Symbol, Type)]) :: [(Symbol, Type)] where ... #
Rename a column in the schema.
Equations
| RenameInSchema old new ('(old, a) ': rest) = '(new, a) ': rest | |
| RenameInSchema old new (col ': rest) = col ': RenameInSchema old new rest | |
| RenameInSchema old new ('[] :: [(Symbol, Type)]) = TypeError (('Text "Cannot rename: column '" ':<>: 'Text old) ':<>: 'Text "' not found") :: [(Symbol, Type)] |
type family RenameManyInSchema (pairs :: [(Symbol, Symbol)]) (cols :: [(Symbol, Type)]) :: [(Symbol, Type)] where ... #
Rename multiple columns.
Equations
| RenameManyInSchema ('[] :: [(Symbol, Symbol)]) cols = cols | |
| RenameManyInSchema ('(old, new) ': rest) cols = RenameManyInSchema rest (RenameInSchema old new cols) |
type family RemoveColumn (name :: Symbol) (cols :: [(Symbol, Type)]) :: [(Symbol, Type)] where ... #
Remove a column by name from a schema.
Equations
| RemoveColumn name ('(name, _1) ': rest) = rest | |
| RemoveColumn name (col ': rest) = col ': RemoveColumn name rest | |
| RemoveColumn name ('[] :: [(Symbol, Type)]) = '[] :: [(Symbol, Type)] |
type family Impute (name :: Symbol) (cols :: [(Symbol, Type)]) :: [(Symbol, Type)] where ... #
Unwrap a Maybe from a type after we impute values.
Equations
| Impute name ('(name, Maybe a) ': rest) = '(name, a) ': rest | |
| Impute name ('(name, _1) ': rest) = TypeError (('Text "Column '" ':<>: 'Text name) ':<>: 'Text "' is not of kind Maybe *") :: [(Symbol, Type)] | |
| Impute name (col ': rest) = col ': Impute name rest | |
| Impute name ('[] :: [(Symbol, Type)]) = '[] :: [(Symbol, Type)] |
type family SetColumnType (name :: Symbol) b (cols :: [(Symbol, Type)]) :: [(Symbol, Type)] where ... #
Equations
| SetColumnType name b ('(name, _1) ': rest) = '(name, b) ': rest | |
| SetColumnType name b (col ': rest) = col ': SetColumnType name b rest | |
| SetColumnType name b ('[] :: [(Symbol, Type)]) = TypeError (('Text "Column '" ':<>: 'Text name) ':<>: 'Text "' not found in schema") :: [(Symbol, Type)] |
type family Reverse (xs :: [(Symbol, Type)]) :: [(Symbol, Type)] where ... #
Reverse a type-level list.
type family StripAllMaybe (cols :: [(Symbol, Type)]) :: [(Symbol, Type)] where ... #
Strip Maybe from all columns. Used by filterAllJust.
'("x", (Maybe Double) becomes '("x", Double.))
'("y", Int stays '("y", Int.))
Equations
| StripAllMaybe ('[] :: [(Symbol, Type)]) = '[] :: [(Symbol, Type)] | |
| StripAllMaybe ('(n, Maybe a) ': rest) = '(n, a) ': StripAllMaybe rest | |
| StripAllMaybe ('(n, a) ': rest) = '(n, a) ': StripAllMaybe rest |
type family StripMaybeAt (name :: Symbol) (cols :: [(Symbol, Type)]) :: [(Symbol, Type)] where ... #
Strip Maybe from a single named column. Used by filterJust.
StripMaybeAt "x" '[ '("x", Maybe Double), '("y", Int)]
= '[ '("x", Double), '("y", Int)]
Equations
| StripMaybeAt name ('(name, Maybe a) ': rest) = '(name, a) ': rest | |
| StripMaybeAt name ('(name, a) ': rest) = '(name, a) ': rest | |
| StripMaybeAt name (col ': rest) = col ': StripMaybeAt name rest | |
| StripMaybeAt name ('[] :: [(Symbol, Type)]) = TypeError (('Text "Column '" ':<>: 'Text name) ':<>: 'Text "' not found in schema") :: [(Symbol, Type)] |
type family GroupKeyColumns (keys :: [Symbol]) (cols :: [(Symbol, Type)]) :: [(Symbol, Type)] where ... #
Extract Column entries from a schema whose names appear in keys.
Equations
| GroupKeyColumns keys ('[] :: [(Symbol, Type)]) = '[] :: [(Symbol, Type)] | |
| GroupKeyColumns keys ('(n, a) ': rest) = GroupKeyColumnsHelper (IsElem n keys) n a keys rest |
type family InnerJoinSchema (keys :: [Symbol]) (left :: [(Symbol, Type)]) (right :: [(Symbol, Type)]) :: [(Symbol, Type)] where ... #
Inner join result schema.
Equations
| InnerJoinSchema keys left right = Append (SubsetSchema keys left) (Append (UniqueLeft left (Append keys (ColumnNames right))) (Append (UniqueLeft right (Append keys (ColumnNames left))) (CollidingColumns left right keys))) |
type family LeftJoinSchema (keys :: [Symbol]) (left :: [(Symbol, Type)]) (right :: [(Symbol, Type)]) :: [(Symbol, Type)] where ... #
Left join result schema.
Equations
| LeftJoinSchema keys left right = Append (SubsetSchema keys left) (Append (UniqueLeft left (Append keys (ColumnNames right))) (Append (WrapMaybe (UniqueLeft right (Append keys (ColumnNames left)))) (CollidingColumns left right keys))) |
type family RightJoinSchema (keys :: [Symbol]) (left :: [(Symbol, Type)]) (right :: [(Symbol, Type)]) :: [(Symbol, Type)] where ... #
Right join result schema.
Equations
| RightJoinSchema keys left right = Append (SubsetSchema keys right) (Append (WrapMaybe (UniqueLeft left (Append keys (ColumnNames right)))) (Append (UniqueLeft right (Append keys (ColumnNames left))) (CollidingColumns left right keys))) |
type family FullOuterJoinSchema (keys :: [Symbol]) (left :: [(Symbol, Type)]) (right :: [(Symbol, Type)]) :: [(Symbol, Type)] where ... #
Full outer join result schema.
Equations
| FullOuterJoinSchema keys left right = Append (WrapMaybe (SubsetSchema keys left)) (Append (WrapMaybe (UniqueLeft left (Append keys (ColumnNames right)))) (Append (WrapMaybe (UniqueLeft right (Append keys (ColumnNames left)))) (CollidingColumns left right keys))) |
type family AssertAbsent (name :: Symbol) (cols :: [(Symbol, Type)]) where ... #
Assert that a column name is absent from the schema (for derive/insert).
Equations
| AssertAbsent name cols = AssertAbsentHelper name (HasName name cols) cols |
type family AssertAllPresent (name :: [Symbol]) (cols :: [(Symbol, Type)]) where ... #
Assert that a column name is present in the schema.
Equations
| AssertAllPresent (name ': rest) cols = AssertAllPresentHelper (HasName name cols) name rest cols | |
| AssertAllPresent ('[] :: [Symbol]) cols = () |
type family AssertPresent (name :: Symbol) (cols :: [(Symbol, Type)]) where ... #
Assert that a column name is present in the schema.
Equations
| AssertPresent name cols = AssertPresentHelper name (HasName name cols) cols |
type family AssertDisjoint (left :: [(Symbol, Type)]) (right :: [(Symbol, Type)]) where ... #
Equations
| AssertDisjoint left right = AssertDisjointHelper (SharedNames left right) left right |
type family AssertRealColumn (fn :: Symbol) (name :: Symbol) a where ... #
Emit a readable compile error when the column named name is not a
real-number type, naming the calling function fn, the column, and the type it
actually has. Used by the numeric extractors so a wrong column type reads as a
repairable message rather than a bare No instance for Real ….
Equations
| AssertRealColumn fn name a = AssertRealColumnGo fn name a (IsRealType a) |
type family AllColumnsReal (fn :: Symbol) (cols :: [(Symbol, Type)]) where ... #
Constraint that every column in the schema is a real (numeric), unboxed
type. Lets the whole-frame matrix extractors (toDoubleMatrix and friends) be
total — a non-numeric or nullable column is a compile error (with the offending
column named, via AssertRealColumn), not a runtime Left.
Equations
| AllColumnsReal fn ('[] :: [(Symbol, Type)]) = () | |
| AllColumnsReal fn ('(n, a) ': rest) = (AssertRealColumn fn n a, Real a, Unbox a, AllColumnsReal fn rest) |
type family IsRealType a :: Bool where ... #
Is a a real, unboxed numeric type — i.e. a valid numeric-column element?
Equations
| IsRealType Int = 'True | |
| IsRealType Int8 = 'True | |
| IsRealType Int16 = 'True | |
| IsRealType Int32 = 'True | |
| IsRealType Int64 = 'True | |
| IsRealType Word = 'True | |
| IsRealType Word8 = 'True | |
| IsRealType Word16 = 'True | |
| IsRealType Word32 = 'True | |
| IsRealType Word64 = 'True | |
| IsRealType Double = 'True | |
| IsRealType Float = 'True | |
| IsRealType _1 = 'False |
Constraints
class KnownSchema (cols :: [(Symbol, Type)]) where #
Provides runtime evidence of a schema: a list of (name, TypeRep) pairs.
Methods
schemaEvidence :: [(Text, SomeTypeRep)] #
Instances
| KnownSchema ('[] :: [(Symbol, Type)]) | |
Defined in DataFrame.Typed.Schema Methods schemaEvidence :: [(Text, SomeTypeRep)] # | |
| (KnownSymbol name, Typeable a, Columnable a, KnownSchema rest) => KnownSchema ('(name, a) ': rest) | |
Defined in DataFrame.Typed.Schema Methods schemaEvidence :: [(Text, SomeTypeRep)] # | |
schemaColumnNames :: forall (cols :: [(Symbol, Type)]). KnownSchema cols => [Text] #
The column names a schema declares, in schema order. Pass it to a reader's options to fetch only those columns:
D.readCsvWithOpts
D.defaultReadOptions{D.readColumns = Just (schemaColumnNames @(Schema Customer))}
"customers.csv"
class AllKnownSymbol (names :: [Symbol]) where #
A class that provides a list of Text values for a type-level list of Symbols.
Methods
symbolVals :: [Text] #
Instances
| AllKnownSymbol ('[] :: [Symbol]) | |
Defined in DataFrame.Typed.Schema Methods symbolVals :: [Text] # | |
| (KnownSymbol n, AllKnownSymbol ns) => AllKnownSymbol (n ': ns) | |
Defined in DataFrame.Typed.Schema Methods symbolVals :: [Text] # | |