dataframe-3.5.0.0: A fast, safe, and intuitive DataFrame library.
Copyright(c) 2024 - 2026 Michael Chavinda
LicenseMIT
Maintainermschavinda@gmail.com
Stabilityexperimental
Safe HaskellNone
LanguageHaskell2010

DataFrame.Typed

Description

A type-safe layer over the dataframe library.

This module provides TypedDataFrame, a phantom-typed wrapper around the untyped DataFrame that tracks column names and types at compile time. All operations delegate to the untyped core at runtime; the phantom type is updated at compile time to reflect schema changes.

Key difference from untyped API: TExpr

All expression-taking operations use TExpr (typed expressions) instead of raw Expr. Column references are validated at compile time:

{-# LANGUAGE DataKinds, TypeApplications, TypeOperators #-}
import qualified DataFrame.Typed as T

type People = '[ '("name", Text), '("age", Int)]

main = do
    raw <- D.readCsv "people.csv"
    case T.freeze @People raw of
        Nothing -> putStrLn "Schema mismatch!"
        Just df -> do
            let adults = T.filterWhere (T.col @"age" T..>=. T.lit 18) df
            let names  = T.columnAsList @"name" adults  -- :: [Text]
            print names

Column references like T.col @"age" are checked at compile time — if the column doesn't exist or has the wrong type, you get a type error, not a runtime exception.

filterAllJust tracks Maybe-stripping

df :: TypedDataFrame '[ '("x", Maybe Double), '("y", Int)]
T.filterAllJust df :: TypedDataFrame '[ '("x", Double), '("y", Int)]

Typed aggregation

result = T.aggregate
    ( T.as @"total" (T.sum   (T.col @"salary"))
    . T.as @"count" (T.count (T.col @"salary"))
    )
    (T.groupBy @'["dept"] employees)
Synopsis

Core types

data TypedDataFrame (cols :: [(Symbol, Type)]) #

A phantom-typed wrapper over the untyped DataFrame.

The type parameter cols is a type-level list of '(name, ty) pairs that tracks the schema at compile time. All operations delegate to the untyped core at runtime and update the phantom type at compile time.

Instances

Instances details
Show (TypedDataFrame cols) 
Instance details

Defined in DataFrame.Typed.Types

ToDataFrame (TypedDataFrame cols) 
Instance details

Defined in DataFrame.Typed.Freeze

Eq (TypedDataFrame cols) 
Instance details

Defined in DataFrame.Typed.Types

Methods

(==) :: TypedDataFrame cols -> TypedDataFrame cols -> Bool #

(/=) :: TypedDataFrame cols -> TypedDataFrame cols -> Bool #

data TypedGrouped (keys :: [Symbol]) (cols :: [(Symbol, Type)]) #

A phantom-typed wrapper over GroupedDataFrame.

data These a b #

Inline replacement for Data.These.These to keep dataframe-core free of the these package dependency. Only the three constructors and the derived classes are used internally.

Constructors

This a 
That b 
These a b 

Instances

Instances details
Foldable (These a) 
Instance details

Defined in DataFrame.Internal.Types

Methods

fold :: Monoid m => These a m -> m #

foldMap :: Monoid m => (a0 -> m) -> These a a0 -> m #

foldMap' :: Monoid m => (a0 -> m) -> These a a0 -> m #

foldr :: (a0 -> b -> b) -> b -> These a a0 -> b #

foldr' :: (a0 -> b -> b) -> b -> These a a0 -> b #

foldl :: (b -> a0 -> b) -> b -> These a a0 -> b #

foldl' :: (b -> a0 -> b) -> b -> These a a0 -> b #

foldr1 :: (a0 -> a0 -> a0) -> These a a0 -> a0 #

foldl1 :: (a0 -> a0 -> a0) -> These a a0 -> a0 #

toList :: These a a0 -> [a0] #

null :: These a a0 -> Bool #

length :: These a a0 -> Int #

elem :: Eq a0 => a0 -> These a a0 -> Bool #

maximum :: Ord a0 => These a a0 -> a0 #

minimum :: Ord a0 => These a a0 -> a0 #

sum :: Num a0 => These a a0 -> a0 #

product :: Num a0 => These a a0 -> a0 #

Traversable (These a) 
Instance details

Defined in DataFrame.Internal.Types

Methods

traverse :: Applicative f => (a0 -> f b) -> These a a0 -> f (These a b) #

sequenceA :: Applicative f => These a (f a0) -> f (These a a0) #

mapM :: Monad m => (a0 -> m b) -> These a a0 -> m (These a b) #

sequence :: Monad m => These a (m a0) -> m (These a a0) #

Functor (These a) 
Instance details

Defined in DataFrame.Internal.Types

Methods

fmap :: (a0 -> b) -> These a a0 -> These a b #

(<$) :: a0 -> These a b -> These a a0 #

(Read a, Read b) => Read (These a b) 
Instance details

Defined in DataFrame.Internal.Types

(Show a, Show b) => Show (These a b) 
Instance details

Defined in DataFrame.Internal.Types

Methods

showsPrec :: Int -> These a b -> ShowS #

show :: These a b -> String #

showList :: [These a b] -> ShowS #

(Eq a, Eq b) => Eq (These a b) 
Instance details

Defined in DataFrame.Internal.Types

Methods

(==) :: These a b -> These a b -> Bool #

(/=) :: These a b -> These a b -> Bool #

(Ord a, Ord b) => Ord (These a b) 
Instance details

Defined in DataFrame.Internal.Types

Methods

compare :: These a b -> These a b -> Ordering #

(<) :: These a b -> These a b -> Bool #

(<=) :: These a b -> These a b -> Bool #

(>) :: These a b -> These a b -> Bool #

(>=) :: These a b -> These a b -> Bool #

max :: These a b -> These a b -> These a b #

min :: These a b -> These a b -> These a b #

Typed expressions

newtype TExpr (cols :: [(Symbol, Type)]) a #

A typed expression validated against schema cols, producing values of type a.

Unlike the untyped 'Expr a', a TExpr can only be constructed through type-safe combinators (col, lit, arithmetic operations) that verify column references exist in the schema with the correct type.

Use unTExpr to extract the underlying Expr for delegation to the untyped API.

Constructors

TExpr 

Fields

Instances

Instances details
Fit cfg [Expr Double] => Fit cfg [TExpr cols Double]

The same lift for the unsupervised feature-list inputs.

Instance details

Defined in DataFrame.Model

Associated Types

type ModelOf cfg [TExpr cols Double] 
Instance details

Defined in DataFrame.Model

type ModelOf cfg [TExpr cols Double] = ModelOf cfg [Expr Double]
type FrameReq cfg [TExpr cols Double] 
Instance details

Defined in DataFrame.Model

type FrameReq cfg [TExpr cols Double] = FrameReq cfg [Expr Double]

Methods

fit :: cfg -> [TExpr cols Double] -> FrameFor [TExpr cols Double] -> FitResult (FrameFor [TExpr cols Double]) (ModelOf cfg [TExpr cols Double]) #

Fit cfg (Expr a) => Fit cfg (TExpr cols a)

Lift any model fittable on an untyped target Expr a to a typed target TExpr cols a over a TypedDataFrame cols, returning a schema-tagged Fitted.

Instance details

Defined in DataFrame.Model

Associated Types

type ModelOf cfg (TExpr cols a) 
Instance details

Defined in DataFrame.Model

type ModelOf cfg (TExpr cols a) = ModelOf cfg (Expr a)
type FrameReq cfg (TExpr cols a) 
Instance details

Defined in DataFrame.Model

type FrameReq cfg (TExpr cols a) = FrameReq cfg (Expr a)

Methods

fit :: cfg -> TExpr cols a -> FrameFor (TExpr cols a) -> FitResult (FrameFor (TExpr cols a)) (ModelOf cfg (TExpr cols a)) #

Show a => Show (TExpr cols a)

Shows the underlying expression; the schema phantom is type-level only.

Instance details

Defined in DataFrame.Typed.Types

Methods

showsPrec :: Int -> TExpr cols a -> ShowS #

show :: TExpr cols a -> String #

showList :: [TExpr cols a] -> ShowS #

type FrameReq cfg [TExpr cols Double] 
Instance details

Defined in DataFrame.Model

type FrameReq cfg [TExpr cols Double] = FrameReq cfg [Expr Double]
type ModelOf cfg [TExpr cols Double] 
Instance details

Defined in DataFrame.Model

type ModelOf cfg [TExpr cols Double] = ModelOf cfg [Expr Double]
type FrameReq cfg (TExpr cols a) 
Instance details

Defined in DataFrame.Model

type FrameReq cfg (TExpr cols a) = FrameReq cfg (Expr a)
type ModelOf cfg (TExpr cols a) 
Instance details

Defined in DataFrame.Model

type ModelOf cfg (TExpr cols a) = ModelOf cfg (Expr a)

col :: forall (name :: Symbol) (cols :: [(Symbol, Type)]) a. (KnownSymbol name, a ~ SafeLookup name cols, Columnable a, AssertPresent name cols) => TExpr cols a #

Create a typed column reference. This is the key type-safety entry point.

The column name must exist in cols and its type must match a. Both checks happen at compile time via type families.

salary :: TExpr '[("salary", Double)] Double
salary = col @"salary"

lit :: forall a (cols :: [(Symbol, Type)]). Columnable a => a -> TExpr cols a #

Create a literal expression. Valid for any schema since it references no columns.

ifThenElse :: forall a (cols :: [(Symbol, Type)]). Columnable a => TExpr cols Bool -> TExpr cols a -> TExpr cols a -> TExpr cols a #

Conditional expression.

lift :: forall a b (cols :: [(Symbol, Type)]). (Columnable a, Columnable b) => (a -> b) -> TExpr cols a -> TExpr cols b #

Lift a unary function into a typed expression.

lift2 :: forall a b c (cols :: [(Symbol, Type)]). (Columnable a, Columnable b, Columnable c) => (a -> b -> c) -> TExpr cols a -> TExpr cols b -> TExpr cols c #

Lift a binary function into typed expressions.

nullLift :: forall a r (cols :: [(Symbol, Type)]). (NullLift1Op a r (NullLift1Result a r), Columnable (NullLift1Result a r)) => (BaseType a -> r) -> TExpr cols a -> TExpr cols (NullLift1Result a r) #

Typed nullLift: lift a unary function with nullable propagation. When the input is Maybe a, Nothing short-circuits; when plain a, applies directly. The return type is inferred via NullLift1Result: no annotation needed.

nullLift2 :: forall a b r (cols :: [(Symbol, Type)]). (NullLift2Op a b r (NullLift2Result a b r), Columnable (NullLift2Result a b r)) => (BaseType a -> BaseType b -> r) -> TExpr cols a -> TExpr cols b -> TExpr cols (NullLift2Result a b r) #

Typed nullLift2: lift a binary function with nullable propagation. Any Nothing operand short-circuits to Nothing in the result. The return type is inferred via NullLift2Result: no annotation needed.

Same-type comparison operators

(.==.) :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Eq a) => TExpr cols a -> TExpr cols a -> TExpr cols Bool infixl 4 #

(./=.) :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Eq a) => TExpr cols a -> TExpr cols a -> TExpr cols Bool infixl 4 #

(.<.) :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TExpr cols a -> TExpr cols Bool infixl 4 #

(.<=.) :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TExpr cols a -> TExpr cols Bool infixl 4 #

(.>=.) :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TExpr cols a -> TExpr cols Bool infixl 4 #

(.>.) :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TExpr cols a -> TExpr cols Bool infixl 4 #

Nullable-aware arithmetic operators

(.+) :: forall a b (cols :: [(Symbol, Type)]). (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b (Promote (BaseType a) (BaseType b)) (WidenResult a b), Num (Promote (BaseType a) (BaseType b))) => TExpr cols a -> TExpr cols b -> TExpr cols (WidenResult a b) infixl 6 #

Nullable-aware addition. Works for all combinations of nullable/non-nullable operands. col @"x" .+ col @"y" -- :: TExpr cols (Maybe Int) when y :: Maybe Int

(.-) :: forall a b (cols :: [(Symbol, Type)]). (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b (Promote (BaseType a) (BaseType b)) (WidenResult a b), Num (Promote (BaseType a) (BaseType b))) => TExpr cols a -> TExpr cols b -> TExpr cols (WidenResult a b) infixl 6 #

Nullable-aware subtraction.

(.*) :: forall a b (cols :: [(Symbol, Type)]). (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b (Promote (BaseType a) (BaseType b)) (WidenResult a b), Num (Promote (BaseType a) (BaseType b))) => TExpr cols a -> TExpr cols b -> TExpr cols (WidenResult a b) infixl 7 #

Nullable-aware multiplication.

(./) :: forall a b (cols :: [(Symbol, Type)]). (DivWidenOp (BaseType a) (BaseType b), NullLift2Op a b (PromoteDiv (BaseType a) (BaseType b)) (WidenResultDiv a b), Fractional (PromoteDiv (BaseType a) (BaseType b))) => TExpr cols a -> TExpr cols b -> TExpr cols (WidenResultDiv a b) infixl 7 #

Nullable-aware division. Integral operands are promoted to Double.

Nullable-aware comparison operators (three-valued logic)

(.==) :: forall a b (cols :: [(Symbol, Type)]). (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Eq (Promote (BaseType a) (BaseType b))) => TExpr cols a -> TExpr cols b -> TExpr cols (NullCmpResult a b) infix 4 #

Nullable-aware equality. Widens numeric operands to their common type, so TExpr cols Double .== TExpr cols Int typechecks. Returns Maybe Bool when either operand is nullable.

(./=) :: forall a b (cols :: [(Symbol, Type)]). (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Eq (Promote (BaseType a) (BaseType b))) => TExpr cols a -> TExpr cols b -> TExpr cols (NullCmpResult a b) infix 4 #

Nullable-aware inequality. Widens numeric operands to their common type.

(.<) :: forall a b (cols :: [(Symbol, Type)]). (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Ord (Promote (BaseType a) (BaseType b))) => TExpr cols a -> TExpr cols b -> TExpr cols (NullCmpResult a b) infix 4 #

Nullable-aware less-than. Widens numeric operands to their common type.

(.<=) :: forall a b (cols :: [(Symbol, Type)]). (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Ord (Promote (BaseType a) (BaseType b))) => TExpr cols a -> TExpr cols b -> TExpr cols (NullCmpResult a b) infix 4 #

Nullable-aware less-than-or-equal. Widens numeric operands to their common type.

(.>=) :: forall a b (cols :: [(Symbol, Type)]). (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Ord (Promote (BaseType a) (BaseType b))) => TExpr cols a -> TExpr cols b -> TExpr cols (NullCmpResult a b) infix 4 #

Nullable-aware greater-than-or-equal. Widens numeric operands to their common type.

(.>) :: forall a b (cols :: [(Symbol, Type)]). (NumericWidenOp (BaseType a) (BaseType b), NullLift2Op a b Bool (NullCmpResult a b), Ord (Promote (BaseType a) (BaseType b))) => TExpr cols a -> TExpr cols b -> TExpr cols (NullCmpResult a b) infix 4 #

Nullable-aware greater-than. Widens numeric operands to their common type.

Logical operators

(.&&.) :: forall (cols :: [(Symbol, Type)]). TExpr cols Bool -> TExpr cols Bool -> TExpr cols Bool infixr 3 #

(.||.) :: forall (cols :: [(Symbol, Type)]). TExpr cols Bool -> TExpr cols Bool -> TExpr cols Bool infixr 2 #

not :: forall (cols :: [(Symbol, Type)]). TExpr cols Bool -> TExpr cols Bool #

Aggregation expression combinators

sum :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Num a) => TExpr cols a -> TExpr cols a #

mean :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a) => TExpr cols a -> TExpr cols Double #

median :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a, Unbox a) => TExpr cols a -> TExpr cols Double #

count :: forall a (cols :: [(Symbol, Type)]). Columnable a => TExpr cols a -> TExpr cols Int #

countAll :: forall (cols :: [(Symbol, Type)]). TExpr cols Int #

Row count, the equivalent of SQL's COUNT(*).

minimum :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TExpr cols a #

maximum :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TExpr cols a #

collect :: forall a (cols :: [(Symbol, Type)]). Columnable a => TExpr cols a -> TExpr cols [a] #

over :: forall (names :: [Symbol]) (cols :: [(Symbol, Type)]) a. (Columnable a, AllKnownSymbol names, AssertAllPresent names cols) => TExpr cols a -> TExpr cols a #

Expression combinators (full DataFrame.Functions parity)

div :: forall a (cols :: [(Symbol, Type)]). (Integral a, Columnable a) => TExpr cols a -> TExpr cols a -> TExpr cols a #

Integer division.

mod :: forall a (cols :: [(Symbol, Type)]). (Integral a, Columnable a) => TExpr cols a -> TExpr cols a -> TExpr cols a #

Integer modulus.

mode :: forall a (cols :: [(Symbol, Type)]). (Ord a, Columnable a, Eq a) => TExpr cols a -> TExpr cols a #

Most frequent value (aggregation).

sumMaybe :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Num a) => TExpr cols (Maybe a) -> TExpr cols a #

Sum of a nullable column, ignoring Nothing (aggregation).

meanMaybe :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a) => TExpr cols (Maybe a) -> TExpr cols Double #

Mean of a nullable column, ignoring Nothing (aggregation).

variance :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a, Unbox a) => TExpr cols a -> TExpr cols Double #

Variance (aggregation).

medianMaybe :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a) => TExpr cols (Maybe a) -> TExpr cols Double #

Median of a nullable column, ignoring Nothing (aggregation).

percentile :: forall (cols :: [(Symbol, Type)]). Int -> TExpr cols Double -> TExpr cols Double #

The n-th percentile (aggregation).

stddev :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a, Unbox a) => TExpr cols a -> TExpr cols Double #

Standard deviation (aggregation).

stddevMaybe :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a) => TExpr cols (Maybe a) -> TExpr cols Double #

Standard deviation of a nullable column, ignoring Nothing (aggregation).

zScore :: forall (cols :: [(Symbol, Type)]). TExpr cols Double -> TExpr cols Double #

Z-score (value minus group mean, over standard deviation).

pow :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Num a) => TExpr cols a -> Int -> TExpr cols a #

Raise an expression to an integer power.

relu :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Num a, Ord a) => TExpr cols a -> TExpr cols a #

Rectified linear unit: max 0.

min :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TExpr cols a -> TExpr cols a #

Element-wise minimum of two expressions.

max :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TExpr cols a -> TExpr cols a #

Element-wise maximum of two expressions.

reduce :: forall a b (cols :: [(Symbol, Type)]). (Columnable a, Columnable b) => TExpr cols b -> a -> (a -> b -> a) -> TExpr cols a #

Fold a column into a single value with a seed and step function (aggregation).

toMaybe :: forall a (cols :: [(Symbol, Type)]). Columnable a => TExpr cols a -> TExpr cols (Maybe a) #

Wrap each value in Just.

fromMaybe :: forall a (cols :: [(Symbol, Type)]). Columnable a => a -> TExpr cols (Maybe a) -> TExpr cols a #

Replace Nothing with a default.

isJust :: forall a (cols :: [(Symbol, Type)]). Columnable a => TExpr cols (Maybe a) -> TExpr cols Bool #

True where the value is Just.

isNothing :: forall a (cols :: [(Symbol, Type)]). Columnable a => TExpr cols (Maybe a) -> TExpr cols Bool #

True where the value is Nothing.

fromJust :: forall a (cols :: [(Symbol, Type)]). Columnable a => TExpr cols (Maybe a) -> TExpr cols a #

Unwrap a Just, erroring on Nothing.

whenPresent :: forall a b (cols :: [(Symbol, Type)]). (Columnable a, Columnable b) => (a -> b) -> TExpr cols (Maybe a) -> TExpr cols (Maybe b) #

Apply a function only where the value is present.

whenBothPresent :: forall a b c (cols :: [(Symbol, Type)]). (Columnable a, Columnable b, Columnable c) => (a -> b -> c) -> TExpr cols (Maybe a) -> TExpr cols (Maybe b) -> TExpr cols (Maybe c) #

Apply a binary function only where both values are present.

recode :: forall a b (cols :: [(Symbol, Type)]). (Columnable a, Columnable b, Show a, Show b, Show (a, b)) => [(a, b)] -> TExpr cols a -> TExpr cols (Maybe b) #

Map values through a lookup table, yielding Nothing for misses.

recodeWithCondition :: forall a b (cols :: [(Symbol, Type)]). (Columnable a, Columnable b) => TExpr cols b -> [(TExpr cols a -> TExpr cols Bool, b)] -> TExpr cols a -> TExpr cols b #

Pick the first value whose condition holds, else a fallback.

recodeWithDefault :: forall a b (cols :: [(Symbol, Type)]). (Columnable a, Columnable b, Show (a, b)) => b -> [(a, b)] -> TExpr cols a -> TExpr cols b #

Map values through a lookup table, with a default for misses.

firstOrNothing :: forall a (cols :: [(Symbol, Type)]). Columnable a => TExpr cols [a] -> TExpr cols (Maybe a) #

First element of a list column, or Nothing.

lastOrNothing :: forall a (cols :: [(Symbol, Type)]). Columnable a => TExpr cols [a] -> TExpr cols (Maybe a) #

Last element of a list column, or Nothing.

splitOn :: forall (cols :: [(Symbol, Type)]). Text -> TExpr cols Text -> TExpr cols [Text] #

Split text on a delimiter.

match :: forall (cols :: [(Symbol, Type)]). Text -> TExpr cols Text -> TExpr cols (Maybe Text) #

First regex match, or Nothing.

matchAll :: forall (cols :: [(Symbol, Type)]). Text -> TExpr cols Text -> TExpr cols [Text] #

All regex matches.

parseDate :: forall t (cols :: [(Symbol, Type)]). (ParseTime t, Columnable t) => Text -> TExpr cols Text -> TExpr cols (Maybe t) #

Parse text into a time value with the given format.

daysBetween :: forall (cols :: [(Symbol, Type)]). TExpr cols Day -> TExpr cols Day -> TExpr cols Int #

Number of days between two dates.

bind :: forall a m b (cols :: [(Symbol, Type)]). (Columnable a, Columnable (m a), Monad m, Columnable b, Columnable (m b)) => (a -> m b) -> TExpr cols (m a) -> TExpr cols (m b) #

Monadic bind over a column of monadic values.

Cast / coercion expressions

castExpr :: forall b (cols :: [(Symbol, Type)]) src. (Columnable b, Columnable src, Read b) => TExpr cols src -> TExpr cols (Maybe b) #

castExprWithDefault :: forall b (cols :: [(Symbol, Type)]) src. (Columnable b, Columnable src, Read b) => b -> TExpr cols src -> TExpr cols b #

castExprEither :: forall b (cols :: [(Symbol, Type)]) src. (Columnable b, Columnable src, Read b) => TExpr cols src -> TExpr cols (Either Text b) #

unsafeCastExpr :: forall b (cols :: [(Symbol, Type)]) src. (Columnable b, Columnable src, Read b) => TExpr cols src -> TExpr cols b #

toDouble :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a) => TExpr cols a -> TExpr cols Double #

Typed sort orders

data TSortOrder (cols :: [(Symbol, Type)]) where #

A typed sort order validated against schema cols.

Constructors

Asc :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TSortOrder cols 
Desc :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TSortOrder cols 

asc :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TSortOrder cols #

Create an ascending sort order from a typed expression.

desc :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TSortOrder cols #

Create a descending sort order from a typed expression.

Freeze / thaw boundary

freeze :: forall (cols :: [(Symbol, Type)]). KnownSchema cols => DataFrame -> Maybe (TypedDataFrame cols) #

Validate that an untyped DataFrame matches the expected schema cols, then wrap it. Returns Nothing on mismatch.

freezeWithError :: forall (cols :: [(Symbol, Type)]). KnownSchema cols => DataFrame -> Either Text (TypedDataFrame cols) #

Like freeze but returns a descriptive error message on failure.

thaw :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> DataFrame #

Unwrap a typed DataFrame back to the untyped representation. Always safe; discards type information.

unsafeFreeze :: forall (cols :: [(Symbol, Type)]). DataFrame -> TypedDataFrame cols #

Wrap an untyped DataFrame without any validation. Used internally after delegation where the library guarantees schema correctness.

Typed column access

columnAsVector :: forall (name :: Symbol) (cols :: [(Symbol, Type)]) a. (KnownSymbol name, a ~ SafeLookup name cols, Columnable a, AssertPresent name cols) => TypedDataFrame cols -> Vector a #

Retrieve a column as a boxed Vector, with the type determined by the schema. The column must exist (enforced at compile time).

columnAsList :: forall (name :: Symbol) (cols :: [(Symbol, Type)]) a. (KnownSymbol name, a ~ SafeLookup name cols, Columnable a, AssertPresent name cols) => TypedDataFrame cols -> [a] #

Retrieve a column as a list, with the type determined by the schema.

columnAsIntVector :: forall (name :: Symbol) (cols :: [(Symbol, Type)]) a. (KnownSymbol name, a ~ SafeLookup name cols, Columnable a, AssertRealColumn "columnAsIntVector" name a, Real a, Unbox a, AssertPresent name cols) => TypedDataFrame cols -> Vector Int #

Retrieve a column coerced to an unboxed Int vector, named by type application. The column must exist and be numeric — both are compile-time checks via SafeLookup, so this is total (no Either, no runtime throw).

columnAsDoubleVector :: forall (name :: Symbol) (cols :: [(Symbol, Type)]) a. (KnownSymbol name, a ~ SafeLookup name cols, Columnable a, AssertRealColumn "columnAsDoubleVector" name a, Real a, Unbox a, AssertPresent name cols) => TypedDataFrame cols -> Vector Double #

Retrieve a column coerced to an unboxed Double vector. See columnAsIntVector.

columnAsFloatVector :: forall (name :: Symbol) (cols :: [(Symbol, Type)]) a. (KnownSymbol name, a ~ SafeLookup name cols, Columnable a, AssertRealColumn "columnAsFloatVector" name a, Real a, Unbox a, AssertPresent name cols) => TypedDataFrame cols -> Vector Float #

Retrieve a column coerced to an unboxed Float vector. See columnAsIntVector.

columnAsUnboxedVector :: forall (name :: Symbol) (cols :: [(Symbol, Type)]) a. (KnownSymbol name, a ~ SafeLookup name cols, Columnable a, Unbox a, AssertPresent name cols) => TypedDataFrame cols -> Vector a #

Retrieve a column as an unboxed vector of its own element type. The column must exist and be unboxable — both compile-time checks, so this is total.

toDoubleMatrix :: forall (cols :: [(Symbol, Type)]). AllColumnsReal "toDoubleMatrix" cols => TypedDataFrame cols -> Vector (Vector Double) #

Convert every column to Double and transpose into a row-major matrix. Total: AllColumnsReal proves at compile time that every column is numeric and unboxed, so the conversion cannot fail.

toFloatMatrix :: forall (cols :: [(Symbol, Type)]). AllColumnsReal "toFloatMatrix" cols => TypedDataFrame cols -> Vector (Vector Float) #

Convert every column to Float and transpose into a row-major matrix. See toDoubleMatrix.

toIntMatrix :: forall (cols :: [(Symbol, Type)]). AllColumnsReal "toIntMatrix" cols => TypedDataFrame cols -> Vector (Vector Int) #

Convert every column to Int and transpose into a row-major matrix. See toDoubleMatrix.

Schema-preserving operations

filterWhere :: forall (cols :: [(Symbol, Type)]). TExpr cols Bool -> TypedDataFrame cols -> TypedDataFrame cols #

Filter rows where a boolean expression evaluates to True. The expression is validated against the schema at compile time.

filter :: forall a (cols :: [(Symbol, Type)]). Columnable a => TExpr cols a -> (a -> Bool) -> TypedDataFrame cols -> TypedDataFrame cols #

Filter rows by applying a predicate to a typed expression.

filterBy :: forall a (cols :: [(Symbol, Type)]). Columnable a => (a -> Bool) -> TExpr cols a -> TypedDataFrame cols -> TypedDataFrame cols #

Filter rows by a predicate on a column expression (flipped argument order).

filterAllJust :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> TypedDataFrame (StripAllMaybe cols) #

Keep only rows where ALL Optional columns have Just values. Strips Maybe from all column types in the result schema.

df :: TDF '[ '("x", Maybe Double), '("y", Int)]
filterAllJust df :: TDF '[ '("x", Double), '("y", Int)]

filterJust :: forall (name :: Symbol) (cols :: [(Symbol, Type)]). (KnownSymbol name, AssertPresent name cols) => TypedDataFrame cols -> TypedDataFrame (StripMaybeAt name cols) #

Keep only rows where the named column has Just values. Strips Maybe from that column's type in the result schema.

filterJust @"x" df

filterNothing :: forall (name :: Symbol) (cols :: [(Symbol, Type)]). (KnownSymbol name, AssertPresent name cols) => TypedDataFrame cols -> TypedDataFrame cols #

Keep only rows where the named column has Nothing. Schema is preserved (column types unchanged, just fewer rows).

filterAllNothing :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> TypedDataFrame cols #

Keep only rows where every nullable column has Nothing. Schema is preserved.

sortBy :: forall (cols :: [(Symbol, Type)]). [TSortOrder cols] -> TypedDataFrame cols -> TypedDataFrame cols #

Sort by the given typed sort orders. Sort orders reference columns that are validated against the schema.

take :: forall (cols :: [(Symbol, Type)]). Int -> TypedDataFrame cols -> TypedDataFrame cols #

Take the first n rows.

takeLast :: forall (cols :: [(Symbol, Type)]). Int -> TypedDataFrame cols -> TypedDataFrame cols #

Take the last n rows.

drop :: forall (cols :: [(Symbol, Type)]). Int -> TypedDataFrame cols -> TypedDataFrame cols #

Drop the first n rows.

dropLast :: forall (cols :: [(Symbol, Type)]). Int -> TypedDataFrame cols -> TypedDataFrame cols #

Drop the last n rows.

range :: forall (cols :: [(Symbol, Type)]). (Int, Int) -> TypedDataFrame cols -> TypedDataFrame cols #

Take rows in the given range (start, end).

cube :: forall (cols :: [(Symbol, Type)]). (Int, Int) -> TypedDataFrame cols -> TypedDataFrame cols #

Take a sub-cube of the DataFrame.

distinct :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> TypedDataFrame cols #

Remove duplicate rows.

sample :: forall g (cols :: [(Symbol, Type)]). RandomGen g => g -> Double -> TypedDataFrame cols -> TypedDataFrame cols #

Randomly sample a fraction of rows.

shuffle :: forall g (cols :: [(Symbol, Type)]). RandomGen g => g -> TypedDataFrame cols -> TypedDataFrame cols #

Shuffle all rows randomly.

Schema-modifying operations

derive :: forall (name :: Symbol) a (cols :: [(Symbol, Type)]). (KnownSymbol name, Columnable a, AssertAbsent name cols) => TExpr cols a -> TypedDataFrame cols -> TypedDataFrame (Snoc cols '(name, a)) #

Derive a new column from a typed expression. The column name must NOT already exist in the schema (enforced at compile time via AssertAbsent). The expression is validated against the current schema.

df' = derive @"total" (col @"price" * col @"qty") df
-- df' :: TDF ('("total", Double ': originalCols))

impute :: forall (name :: Symbol) a (cols :: [(Symbol, Type)]). (KnownSymbol name, Columnable a, Maybe a ~ Lookup name cols) => a -> TypedDataFrame cols -> TypedDataFrame (Impute name cols) #

select :: forall (names :: [Symbol]) (cols :: [(Symbol, Type)]). (AllKnownSymbol names, AssertAllPresent names cols) => TypedDataFrame cols -> TypedDataFrame (SubsetSchema names cols) #

Select a subset of columns by name.

exclude :: forall (names :: [Symbol]) (cols :: [(Symbol, Type)]). AllKnownSymbol names => TypedDataFrame cols -> TypedDataFrame (ExcludeSchema names cols) #

Exclude columns by name.

rename :: forall (old :: Symbol) (new :: Symbol) (cols :: [(Symbol, Type)]). (KnownSymbol old, KnownSymbol new) => TypedDataFrame cols -> TypedDataFrame (RenameInSchema old new cols) #

Rename a column.

renameMany :: forall (pairs :: [(Symbol, Symbol)]) (cols :: [(Symbol, Type)]). AllKnownPairs pairs => TypedDataFrame cols -> TypedDataFrame (RenameManyInSchema pairs cols) #

Rename multiple columns from a type-level list of pairs.

insert :: forall (name :: Symbol) a (cols :: [(Symbol, Type)]) t. (KnownSymbol name, Columnable a, Foldable t, AssertAbsent name cols) => t a -> TypedDataFrame cols -> TypedDataFrame ('(name, a) ': cols) #

Insert a new column from a Foldable container.

insertColumn :: forall (name :: Symbol) a (cols :: [(Symbol, Type)]). (KnownSymbol name, Columnable a, AssertAbsent name cols) => Column -> TypedDataFrame cols -> TypedDataFrame ('(name, a) ': cols) #

Insert a raw Column value.

insertVector :: forall (name :: Symbol) a (cols :: [(Symbol, Type)]). (KnownSymbol name, Columnable a, AssertAbsent name cols) => Vector a -> TypedDataFrame cols -> TypedDataFrame ('(name, a) ': cols) #

Insert a boxed Vector.

cloneColumn :: forall (old :: Symbol) (new :: Symbol) (cols :: [(Symbol, Type)]). (KnownSymbol old, KnownSymbol new, AssertPresent old cols, AssertAbsent new cols) => TypedDataFrame cols -> TypedDataFrame ('(new, Lookup old cols) ': cols) #

Clone an existing column under a new name.

dropColumn :: forall (name :: Symbol) (cols :: [(Symbol, Type)]). (KnownSymbol name, AssertPresent name cols) => TypedDataFrame cols -> TypedDataFrame (RemoveColumn name cols) #

Drop a column by name.

replaceColumn :: forall (name :: Symbol) a (cols :: [(Symbol, Type)]). (KnownSymbol name, Columnable a, a ~ SafeLookup name cols, AssertPresent name cols) => TExpr cols a -> TypedDataFrame cols -> TypedDataFrame cols #

Replace an existing column with new values derived from a typed expression. The column must already exist and the new type must match.

Metadata

dimensions :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> (Int, Int) #

nRows :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> Int #

nColumns :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> Int #

columnNames :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> [Text] #

Vertical merge

append :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> TypedDataFrame cols -> TypedDataFrame cols #

Vertically merge two DataFrames with the same schema.

Set algebra (topos operations)

union :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> TypedDataFrame cols -> TypedDataFrame cols #

Rows appearing in either DataFrame, deduplicated (set union).

intersect :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> TypedDataFrame cols -> TypedDataFrame cols #

Rows appearing in both DataFrames, deduplicated (set intersection).

difference :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> TypedDataFrame cols -> TypedDataFrame cols #

Rows in the left DataFrame but not the right, deduplicated (relational EXCEPT; the subobject complement).

symmetricDifference :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> TypedDataFrame cols -> TypedDataFrame cols #

Rows in exactly one of the two DataFrames, deduplicated.

Joins

innerJoin :: forall (keys :: [Symbol]) (left :: [(Symbol, Type)]) (right :: [(Symbol, Type)]). (AllKnownSymbol keys, AssertAllPresent keys left, AssertAllPresent keys right, AssertKeyTypesMatch keys left right) => TypedDataFrame left -> TypedDataFrame right -> TypedDataFrame (InnerJoinSchema keys left right) #

Typed inner join on one or more key columns.

leftJoin :: forall (keys :: [Symbol]) (left :: [(Symbol, Type)]) (right :: [(Symbol, Type)]). (AllKnownSymbol keys, AssertAllPresent keys left, AssertAllPresent keys right, AssertKeyTypesMatch keys left right) => TypedDataFrame left -> TypedDataFrame right -> TypedDataFrame (LeftJoinSchema keys left right) #

Typed left join.

rightJoin :: forall (keys :: [Symbol]) (left :: [(Symbol, Type)]) (right :: [(Symbol, Type)]). (AllKnownSymbol keys, AssertAllPresent keys left, AssertAllPresent keys right, AssertKeyTypesMatch keys left right) => TypedDataFrame left -> TypedDataFrame right -> TypedDataFrame (RightJoinSchema keys left right) #

Typed right join.

fullOuterJoin :: forall (keys :: [Symbol]) (left :: [(Symbol, Type)]) (right :: [(Symbol, Type)]). (AllKnownSymbol keys, AssertAllPresent keys left, AssertAllPresent keys right, AssertKeyTypesMatch keys left right) => TypedDataFrame left -> TypedDataFrame right -> TypedDataFrame (FullOuterJoinSchema keys left right) #

Typed full outer join.

GroupBy and Aggregation

groupBy :: forall (keys :: [Symbol]) (cols :: [(Symbol, Type)]). (AllKnownSymbol keys, AssertAllPresent keys cols) => TypedDataFrame cols -> TypedGrouped keys cols #

Group a typed DataFrame by one or more key columns.

grouped = groupBy @'["department"] employees

as :: forall (name :: Symbol) a (keys :: [Symbol]) (cols :: [(Symbol, Type)]) (aggs :: [(Symbol, Type)]). (KnownSymbol name, Columnable a) => TExpr cols a -> TAgg keys cols aggs -> TAgg keys cols ('(name, a) ': aggs) #

Build a named aggregation entry. The result column name is supplied via TypeApplications; the underlying expression is validated against the source schema at compile time.

as produces a transformer on the aggregation chain — entries compose with plain (.) from Prelude (or via (|>) for SQL-like postfix reading). aggregate applies the composed transformer to the empty chain internally, so no terminator is needed.

Prefix form

Expand
result = grouped |> aggregate
    ( as @"total"  (sum   (col @"amount"))
    . as @"orders" (count (col @"order_id"))
    . as @"avg"    (mean  (col @"amount"))
    )

Postfix form (SQL-like)

Expand
result = grouped |> aggregate
    ( (sum   (col @"amount")   |> as @"total")
    . (count (col @"order_id") |> as @"orders")
    . (mean  (col @"amount")   |> as @"avg")
    )

Per-entry parentheses are required in the postfix form because (.) binds tighter than (|>).

aggregate :: forall (keys :: [Symbol]) (cols :: [(Symbol, Type)]) (aggs :: [(Symbol, Type)]). (TAgg keys cols ('[] :: [(Symbol, Type)]) -> TAgg keys cols aggs) -> TypedGrouped keys cols -> TypedDataFrame (Append (GroupKeyColumns keys cols) (Reverse aggs)) #

Run a typed aggregation against a grouped DataFrame.

The first argument is a chain of as entries composed with (.). The empty composition (id) yields just the group keys. The result schema is the group-key columns followed by the aggregation columns in declaration order.

result = grouped |> aggregate
    ( as @"total"  (sum (col @"amount"))
    . as @"orders" (count (col @"order_id"))
    )
-- result :: TypedDataFrame
--     '[ '("region", Text)
--      , '("total", Double)
--      , '("orders", Int)
--      ]

aggregateUntyped :: forall (keys :: [Symbol]) (cols :: [(Symbol, Type)]). [NamedExpr] -> TypedGrouped keys cols -> DataFrame #

Escape hatch: run an untyped aggregation and return a raw DataFrame.

Column transformations

applyColumn :: forall (name :: Symbol) a b (cols :: [(Symbol, Type)]). (KnownSymbol name, a ~ SafeLookup name cols, Columnable a, Columnable b, AssertPresent name cols) => (a -> b) -> TypedDataFrame cols -> TypedDataFrame (SetColumnType name b cols) #

Map a function over a column, rewriting its element type from a to b. The schema's entry for name is updated via SetColumnType.

df' = applyColumn @"age" (show :: Int -> String) df
-- the "age" column is now String-typed

applyMany :: forall (names :: [Symbol]) a (cols :: [(Symbol, Type)]). (AllKnownSymbol names, Columnable a, AssertAllColumnsHaveType names a cols) => (a -> a) -> TypedDataFrame cols -> TypedDataFrame cols #

Apply a type-preserving function to several columns at once. Every named column must already share the element type a (enforced by AssertAllColumnsHaveType).

applyWhere :: forall (filterName :: Symbol) (targetName :: Symbol) a b (cols :: [(Symbol, Type)]). (KnownSymbol filterName, KnownSymbol targetName, a ~ SafeLookup filterName cols, b ~ SafeLookup targetName cols, Columnable a, Columnable b, AssertPresent filterName cols, AssertPresent targetName cols) => (a -> Bool) -> (b -> b) -> TypedDataFrame cols -> TypedDataFrame cols #

Apply a function to a target column only on rows where a condition holds on a filter column. Both columns are named by type application; the target keeps its type.

applyWhere @"flagged" @"score" id (* 2) df

applyAtIndex :: forall (name :: Symbol) a (cols :: [(Symbol, Type)]). (KnownSymbol name, a ~ SafeLookup name cols, Columnable a, AssertPresent name cols) => Int -> (a -> a) -> TypedDataFrame cols -> TypedDataFrame cols #

Apply a type-preserving function to a single row of a column.

safeApply :: forall (name :: Symbol) a b (cols :: [(Symbol, Type)]). (KnownSymbol name, a ~ SafeLookup name cols, Columnable a, Columnable b, AssertPresent name cols) => (a -> b) -> TypedDataFrame cols -> Either DataFrameException (TypedDataFrame (SetColumnType name b cols)) #

Like applyColumn but returns the error instead of throwing.

deriveWithExpr :: forall (name :: Symbol) a (cols :: [(Symbol, Type)]). (KnownSymbol name, Columnable a, AssertAbsent name cols) => TExpr cols a -> TypedDataFrame cols -> (TExpr (Snoc cols '(name, a)) a, TypedDataFrame (Snoc cols '(name, a))) #

Derive a new column and also return a typed reference to it. The returned expression lives in the extended schema, so it can feed later operations.

insertWithDefault :: forall (name :: Symbol) a (cols :: [(Symbol, Type)]) t. (KnownSymbol name, Columnable a, Foldable t, AssertAbsent name cols) => a -> t a -> TypedDataFrame cols -> TypedDataFrame ('(name, a) ': cols) #

Insert a column from a Foldable, padding missing rows with a default.

insertVectorWithDefault :: forall (name :: Symbol) a (cols :: [(Symbol, Type)]). (KnownSymbol name, Columnable a, AssertAbsent name cols) => a -> Vector a -> TypedDataFrame cols -> TypedDataFrame ('(name, a) ': cols) #

Insert a boxed Vector, padding missing rows with a default.

insertUnboxedVector :: forall (name :: Symbol) a (cols :: [(Symbol, Type)]). (KnownSymbol name, Columnable a, Unbox a, AssertAbsent name cols) => Vector a -> TypedDataFrame cols -> TypedDataFrame ('(name, a) ': cols) #

Insert an unboxed Vector as a new column.

(|||) :: forall (left :: [(Symbol, Type)]) (right :: [(Symbol, Type)]). AssertDisjoint left right => TypedDataFrame left -> TypedDataFrame right -> TypedDataFrame (Append left right) #

Horizontal merge: place two DataFrames side by side. The schemas must be disjoint (no shared column names), enforced by AssertDisjoint; the result schema is their concatenation.

Sampling and splitting

randomSplit :: forall g (cols :: [(Symbol, Type)]). RandomGen g => g -> Double -> TypedDataFrame cols -> (TypedDataFrame cols, TypedDataFrame cols) #

Split rows into two DataFrames by a fraction.

kFolds :: forall g (cols :: [(Symbol, Type)]). RandomGen g => g -> Int -> TypedDataFrame cols -> [TypedDataFrame cols] #

Partition rows into k folds.

selectRows :: forall (cols :: [(Symbol, Type)]). [Int] -> TypedDataFrame cols -> TypedDataFrame cols #

Select rows by index. | This may fail if the indices are out of bounds; | use with caution or use filter to select rows by a predicate instead.

stratifiedSample :: forall g a (cols :: [(Symbol, Type)]). (SplittableGen g, Columnable a) => g -> Double -> TExpr cols a -> TypedDataFrame cols -> TypedDataFrame cols #

Sample a fraction of rows, preserving the distribution of a strata column.

stratifiedSplit :: forall g a (cols :: [(Symbol, Type)]). (SplittableGen g, Columnable a) => g -> Double -> TExpr cols a -> TypedDataFrame cols -> (TypedDataFrame cols, TypedDataFrame cols) #

Split rows by a fraction, preserving the distribution of a strata column.

Frequencies

valueCounts :: forall a (cols :: [(Symbol, Type)]). (Ord a, Columnable a) => TExpr cols a -> TypedDataFrame cols -> [(a, Int)] #

Count occurrences of each distinct value in a column.

valueProportions :: forall a (cols :: [(Symbol, Type)]). (Ord a, Columnable a) => TExpr cols a -> TypedDataFrame cols -> [(a, Double)] #

Proportion of each distinct value in a column.

Template Haskell

deriveSchemaFromParquetFile :: String -> String -> DecsQ #

Derive a typed schema synonym from a Parquet file (or directory/glob).

data SchemaOptions #

Options controlling deriveSchemaFromTypeWith.

Constructors

SchemaOptions 

Fields

Record bridge (ADT - TypedDataFrame)

class HasSchema a where #

Bridge a Haskell record type a to a typed-dataframe schema.

The schema is exposed as an associated type family Schema so that instances can pick it up from a Rep computation (see SchemaOf) or from an explicit list emitted by deriveSchemaFromType.

toColumns explodes a list of records into a list of named columns. fromColumns reconstructs the records from a DataFrame, returning Left err if a column is missing or has the wrong type.

Associated Types

type Schema a :: [(Symbol, Type)] #

Methods

toColumns :: [a] -> [(Text, Column)] #

fromColumns :: DataFrame -> Either Text [a] #

fromRecordsTyped :: HasSchema a => [a] -> TypedDataFrame (Schema a) #

Like fromRecords but returns a TypedDataFrame tagged with the schema.

Generics opt-in for schema derivation

type SchemaOf a = RepToSchema 'SnakeCase (Rep a) #

Snake_case schema derived from a's Generic representation.

type SchemaOfRaw a = RepToSchema 'IdentityCase (Rep a) #

Identity-cased schema derived from a's Generic representation.

data NameCase #

Field-name policy applied to record selectors when computing RepToSchema.

  • SnakeCase — translate camelCaseField to "camel_case_field".
  • IdentityCase — keep the selector name verbatim.

Constructors

SnakeCase 
IdentityCase 

genericToColumns :: (Generic a, GHasColumns (Rep a)) => [a] -> [(Text, Column)] #

Default implementation of toColumns for any Generic record. Field names are translated with camelCase -> snake_case.

instance HasSchema Order (SchemaOf Order) where
  toColumns   = genericToColumns
  fromColumns = genericFromColumns

genericFromColumns :: (Generic a, GHasColumns (Rep a)) => DataFrame -> Either Text [a] #

Default implementation of fromColumns for any Generic record.

Schema type families (for advanced use)

type family Lookup (name :: Symbol) (cols :: [(Symbol, Type)]) where ... #

Look up the element type of a column by name.

Equations

Lookup name ('(name, a) ': _1) = a 
Lookup name (_1 ': rest) = Lookup name rest 
Lookup name ('[] :: [(Symbol, Type)]) = TypeError (('Text "Column '" ':<>: 'Text name) ':<>: 'Text "' not found in schema") :: Type 

type family SafeLookup (name :: Symbol) (cols :: [(Symbol, Type)]) where ... #

Like Lookup, but returns a harmless fallback (Int) instead of TypeError when the column is not found. Use together with AssertPresent so the error fires exactly once.

Equations

SafeLookup name ('(name, a) ': _1) = a 
SafeLookup name (_1 ': rest) = SafeLookup name rest 
SafeLookup name ('[] :: [(Symbol, Type)]) = Int 

type family HasName (name :: Symbol) (cols :: [(Symbol, Type)]) :: Bool where ... #

Check whether a column name exists in a schema (type-level Bool).

Equations

HasName name ('(name, _1) ': _2) = 'True 
HasName name (_1 ': rest) = HasName name rest 
HasName name ('[] :: [(Symbol, Type)]) = 'False 

type family SubsetSchema (names :: [Symbol]) (cols :: [(Symbol, Type)]) :: [(Symbol, Type)] where ... #

Select a subset of columns by a list of names.

Equations

SubsetSchema ('[] :: [Symbol]) cols = '[] :: [(Symbol, Type)] 
SubsetSchema (n ': ns) cols = '(n, Lookup n cols) ': SubsetSchema ns cols 

type family ExcludeSchema (names :: [Symbol]) (cols :: [(Symbol, Type)]) :: [(Symbol, Type)] where ... #

Exclude columns by a list of names.

Equations

ExcludeSchema names ('[] :: [(Symbol, Type)]) = '[] :: [(Symbol, Type)] 
ExcludeSchema names ('(n, a) ': rest) = ExcludeSchemaHelper (IsElem n names) n a names rest 

type family RenameInSchema (old :: Symbol) (new :: Symbol) (cols :: [(Symbol, Type)]) :: [(Symbol, Type)] where ... #

Rename a column in the schema.

Equations

RenameInSchema old new ('(old, a) ': rest) = '(new, a) ': rest 
RenameInSchema old new (col ': rest) = col ': RenameInSchema old new rest 
RenameInSchema old new ('[] :: [(Symbol, Type)]) = TypeError (('Text "Cannot rename: column '" ':<>: 'Text old) ':<>: 'Text "' not found") :: [(Symbol, Type)] 

type family RenameManyInSchema (pairs :: [(Symbol, Symbol)]) (cols :: [(Symbol, Type)]) :: [(Symbol, Type)] where ... #

Rename multiple columns.

Equations

RenameManyInSchema ('[] :: [(Symbol, Symbol)]) cols = cols 
RenameManyInSchema ('(old, new) ': rest) cols = RenameManyInSchema rest (RenameInSchema old new cols) 

type family RemoveColumn (name :: Symbol) (cols :: [(Symbol, Type)]) :: [(Symbol, Type)] where ... #

Remove a column by name from a schema.

Equations

RemoveColumn name ('(name, _1) ': rest) = rest 
RemoveColumn name (col ': rest) = col ': RemoveColumn name rest 
RemoveColumn name ('[] :: [(Symbol, Type)]) = '[] :: [(Symbol, Type)] 

type family Impute (name :: Symbol) (cols :: [(Symbol, Type)]) :: [(Symbol, Type)] where ... #

Unwrap a Maybe from a type after we impute values.

Equations

Impute name ('(name, Maybe a) ': rest) = '(name, a) ': rest 
Impute name ('(name, _1) ': rest) = TypeError (('Text "Column '" ':<>: 'Text name) ':<>: 'Text "' is not of kind Maybe *") :: [(Symbol, Type)] 
Impute name (col ': rest) = col ': Impute name rest 
Impute name ('[] :: [(Symbol, Type)]) = '[] :: [(Symbol, Type)] 

type family SetColumnType (name :: Symbol) b (cols :: [(Symbol, Type)]) :: [(Symbol, Type)] where ... #

Equations

SetColumnType name b ('(name, _1) ': rest) = '(name, b) ': rest 
SetColumnType name b (col ': rest) = col ': SetColumnType name b rest 
SetColumnType name b ('[] :: [(Symbol, Type)]) = TypeError (('Text "Column '" ':<>: 'Text name) ':<>: 'Text "' not found in schema") :: [(Symbol, Type)] 

type family Append (xs :: [k]) (ys :: [k]) :: [k] where ... #

Append two type-level lists.

Equations

Append ('[] :: [k]) (ys :: [k]) = ys 
Append (x ': xs :: [k]) (ys :: [k]) = x ': Append xs ys 

type family Reverse (xs :: [(Symbol, Type)]) :: [(Symbol, Type)] where ... #

Reverse a type-level list.

Equations

Reverse xs = ReverseAcc xs ('[] :: [(Symbol, Type)]) 

type family StripAllMaybe (cols :: [(Symbol, Type)]) :: [(Symbol, Type)] where ... #

Strip Maybe from all columns. Used by filterAllJust.

'("x", (Maybe Double) becomes '("x", Double.)) '("y", Int stays '("y", Int.))

Equations

StripAllMaybe ('[] :: [(Symbol, Type)]) = '[] :: [(Symbol, Type)] 
StripAllMaybe ('(n, Maybe a) ': rest) = '(n, a) ': StripAllMaybe rest 
StripAllMaybe ('(n, a) ': rest) = '(n, a) ': StripAllMaybe rest 

type family StripMaybeAt (name :: Symbol) (cols :: [(Symbol, Type)]) :: [(Symbol, Type)] where ... #

Strip Maybe from a single named column. Used by filterJust.

StripMaybeAt "x" '[ '("x", Maybe Double), '("y", Int)] = '[ '("x", Double), '("y", Int)]

Equations

StripMaybeAt name ('(name, Maybe a) ': rest) = '(name, a) ': rest 
StripMaybeAt name ('(name, a) ': rest) = '(name, a) ': rest 
StripMaybeAt name (col ': rest) = col ': StripMaybeAt name rest 
StripMaybeAt name ('[] :: [(Symbol, Type)]) = TypeError (('Text "Column '" ':<>: 'Text name) ':<>: 'Text "' not found in schema") :: [(Symbol, Type)] 

type family GroupKeyColumns (keys :: [Symbol]) (cols :: [(Symbol, Type)]) :: [(Symbol, Type)] where ... #

Extract Column entries from a schema whose names appear in keys.

Equations

GroupKeyColumns keys ('[] :: [(Symbol, Type)]) = '[] :: [(Symbol, Type)] 
GroupKeyColumns keys ('(n, a) ': rest) = GroupKeyColumnsHelper (IsElem n keys) n a keys rest 

type family InnerJoinSchema (keys :: [Symbol]) (left :: [(Symbol, Type)]) (right :: [(Symbol, Type)]) :: [(Symbol, Type)] where ... #

Inner join result schema.

Equations

InnerJoinSchema keys left right = Append (SubsetSchema keys left) (Append (UniqueLeft left (Append keys (ColumnNames right))) (Append (UniqueLeft right (Append keys (ColumnNames left))) (CollidingColumns left right keys))) 

type family LeftJoinSchema (keys :: [Symbol]) (left :: [(Symbol, Type)]) (right :: [(Symbol, Type)]) :: [(Symbol, Type)] where ... #

Left join result schema.

Equations

LeftJoinSchema keys left right = Append (SubsetSchema keys left) (Append (UniqueLeft left (Append keys (ColumnNames right))) (Append (WrapMaybe (UniqueLeft right (Append keys (ColumnNames left)))) (CollidingColumns left right keys))) 

type family RightJoinSchema (keys :: [Symbol]) (left :: [(Symbol, Type)]) (right :: [(Symbol, Type)]) :: [(Symbol, Type)] where ... #

Right join result schema.

Equations

RightJoinSchema keys left right = Append (SubsetSchema keys right) (Append (WrapMaybe (UniqueLeft left (Append keys (ColumnNames right)))) (Append (UniqueLeft right (Append keys (ColumnNames left))) (CollidingColumns left right keys))) 

type family FullOuterJoinSchema (keys :: [Symbol]) (left :: [(Symbol, Type)]) (right :: [(Symbol, Type)]) :: [(Symbol, Type)] where ... #

Full outer join result schema.

Equations

FullOuterJoinSchema keys left right = Append (WrapMaybe (SubsetSchema keys left)) (Append (WrapMaybe (UniqueLeft left (Append keys (ColumnNames right)))) (Append (WrapMaybe (UniqueLeft right (Append keys (ColumnNames left)))) (CollidingColumns left right keys))) 

type family AssertAbsent (name :: Symbol) (cols :: [(Symbol, Type)]) where ... #

Assert that a column name is absent from the schema (for derive/insert).

Equations

AssertAbsent name cols = AssertAbsentHelper name (HasName name cols) cols 

type family AssertAllPresent (name :: [Symbol]) (cols :: [(Symbol, Type)]) where ... #

Assert that a column name is present in the schema.

Equations

AssertAllPresent (name ': rest) cols = AssertAllPresentHelper (HasName name cols) name rest cols 
AssertAllPresent ('[] :: [Symbol]) cols = () 

type family AssertPresent (name :: Symbol) (cols :: [(Symbol, Type)]) where ... #

Assert that a column name is present in the schema.

Equations

AssertPresent name cols = AssertPresentHelper name (HasName name cols) cols 

type family AssertDisjoint (left :: [(Symbol, Type)]) (right :: [(Symbol, Type)]) where ... #

Equations

AssertDisjoint left right = AssertDisjointHelper (SharedNames left right) left right 

type family AssertRealColumn (fn :: Symbol) (name :: Symbol) a where ... #

Emit a readable compile error when the column named name is not a real-number type, naming the calling function fn, the column, and the type it actually has. Used by the numeric extractors so a wrong column type reads as a repairable message rather than a bare No instance for Real ….

Equations

AssertRealColumn fn name a = AssertRealColumnGo fn name a (IsRealType a) 

type family AllColumnsReal (fn :: Symbol) (cols :: [(Symbol, Type)]) where ... #

Constraint that every column in the schema is a real (numeric), unboxed type. Lets the whole-frame matrix extractors (toDoubleMatrix and friends) be total — a non-numeric or nullable column is a compile error (with the offending column named, via AssertRealColumn), not a runtime Left.

Equations

AllColumnsReal fn ('[] :: [(Symbol, Type)]) = () 
AllColumnsReal fn ('(n, a) ': rest) = (AssertRealColumn fn n a, Real a, Unbox a, AllColumnsReal fn rest) 

type family IsRealType a :: Bool where ... #

Is a a real, unboxed numeric type — i.e. a valid numeric-column element?

Constraints

class KnownSchema (cols :: [(Symbol, Type)]) where #

Provides runtime evidence of a schema: a list of (name, TypeRep) pairs.

Instances

Instances details
KnownSchema ('[] :: [(Symbol, Type)]) 
Instance details

Defined in DataFrame.Typed.Schema

(KnownSymbol name, Typeable a, Columnable a, KnownSchema rest) => KnownSchema ('(name, a) ': rest) 
Instance details

Defined in DataFrame.Typed.Schema

schemaColumnNames :: forall (cols :: [(Symbol, Type)]). KnownSchema cols => [Text] #

The column names a schema declares, in schema order. Pass it to a reader's options to fetch only those columns:

D.readCsvWithOpts
    D.defaultReadOptions{D.readColumns = Just (schemaColumnNames @(Schema Customer))}
    "customers.csv"

class AllKnownSymbol (names :: [Symbol]) where #

A class that provides a list of Text values for a type-level list of Symbols.

Methods

symbolVals :: [Text] #

Instances

Instances details
AllKnownSymbol ('[] :: [Symbol]) 
Instance details

Defined in DataFrame.Typed.Schema

Methods

symbolVals :: [Text] #

(KnownSymbol n, AllKnownSymbol ns) => AllKnownSymbol (n ': ns) 
Instance details

Defined in DataFrame.Typed.Schema

Methods

symbolVals :: [Text] #