| Safe Haskell | None |
|---|---|
| Language | Haskell2010 |
DataFrame.Typed.Statistics
Description
Typed statistical reducers over a TypedDataFrame.
These mirror the untyped reducers in DataFrame.Operations.Statistics, taking
a schema-checked TExpr instead of a raw Expr. The names (sum, mean,
median, …) deliberately collide with the aggregation-expression combinators
in DataFrame.Typed.Expr, so this module is meant to be imported qualified:
import qualified DataFrame.Typed.Statistics as TS avg = TS.mean (col @"salary") employees
Synopsis
- mean :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a, Unbox a) => TExpr cols a -> TypedDataFrame cols -> Double
- meanMaybe :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a) => TExpr cols (Maybe a) -> TypedDataFrame cols -> Double
- median :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a, Unbox a) => TExpr cols a -> TypedDataFrame cols -> Double
- medianMaybe :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a) => TExpr cols (Maybe a) -> TypedDataFrame cols -> Double
- percentile :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a, Unbox a) => Int -> TExpr cols a -> TypedDataFrame cols -> Double
- genericPercentile :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => Int -> TExpr cols a -> TypedDataFrame cols -> a
- standardDeviation :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a, Unbox a) => TExpr cols a -> TypedDataFrame cols -> Double
- skewness :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a, Unbox a) => TExpr cols a -> TypedDataFrame cols -> Double
- variance :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a, Unbox a) => TExpr cols a -> TypedDataFrame cols -> Double
- interQuartileRange :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a, Unbox a) => TExpr cols a -> TypedDataFrame cols -> Double
- sum :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Num a) => TExpr cols a -> TypedDataFrame cols -> a
- correlation :: forall (c1 :: Symbol) (c2 :: Symbol) a b (cols :: [(Symbol, Type)]). (KnownSymbol c1, KnownSymbol c2, a ~ SafeLookup c1 cols, b ~ SafeLookup c2 cols, Columnable a, Columnable b, Real a, Real b, Unbox a, Unbox b, AssertPresent c1 cols, AssertPresent c2 cols) => TypedDataFrame cols -> Maybe Double
- frequencies :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TypedDataFrame cols -> DataFrame
- imputeWith :: forall a (cols :: [(Symbol, Type)]). (ImputeOp a, Columnable (BaseType a)) => (TExpr cols (BaseType a) -> TExpr cols (BaseType a)) -> TExpr cols a -> TypedDataFrame cols -> TypedDataFrame cols
- summarize :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> DataFrame
- describeColumns :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> DataFrame
Documentation
mean :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a, Unbox a) => TExpr cols a -> TypedDataFrame cols -> Double Source #
Mean of a column.
meanMaybe :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a) => TExpr cols (Maybe a) -> TypedDataFrame cols -> Double Source #
Mean of a nullable column, ignoring Nothing.
median :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a, Unbox a) => TExpr cols a -> TypedDataFrame cols -> Double Source #
Median of a column.
medianMaybe :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a) => TExpr cols (Maybe a) -> TypedDataFrame cols -> Double Source #
Median of a nullable column, ignoring Nothing.
percentile :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a, Unbox a) => Int -> TExpr cols a -> TypedDataFrame cols -> Double Source #
The n-th percentile of a column.
genericPercentile :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => Int -> TExpr cols a -> TypedDataFrame cols -> a Source #
The n-th percentile of a column of any Ord type.
standardDeviation :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a, Unbox a) => TExpr cols a -> TypedDataFrame cols -> Double Source #
Standard deviation of a column.
skewness :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a, Unbox a) => TExpr cols a -> TypedDataFrame cols -> Double Source #
Skewness of a column.
variance :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a, Unbox a) => TExpr cols a -> TypedDataFrame cols -> Double Source #
Variance of a column.
interQuartileRange :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Real a, Unbox a) => TExpr cols a -> TypedDataFrame cols -> Double Source #
Inter-quartile range of a column.
sum :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Num a) => TExpr cols a -> TypedDataFrame cols -> a Source #
Sum of a column.
correlation :: forall (c1 :: Symbol) (c2 :: Symbol) a b (cols :: [(Symbol, Type)]). (KnownSymbol c1, KnownSymbol c2, a ~ SafeLookup c1 cols, b ~ SafeLookup c2 cols, Columnable a, Columnable b, Real a, Real b, Unbox a, Unbox b, AssertPresent c1 cols, AssertPresent c2 cols) => TypedDataFrame cols -> Maybe Double Source #
Pearson's correlation coefficient between two columns, named by type
application. Both columns must exist in the schema and be numeric — these are
checked at compile time via SafeLookup on each name.
TS.correlation @"height" @"weight" people
frequencies :: forall a (cols :: [(Symbol, Type)]). (Columnable a, Ord a) => TExpr cols a -> TypedDataFrame cols -> DataFrame Source #
Frequency table for a column. The result schema is data-dependent
(one column per distinct value), so an untyped DataFrame is returned.
imputeWith :: forall a (cols :: [(Symbol, Type)]). (ImputeOp a, Columnable (BaseType a)) => (TExpr cols (BaseType a) -> TExpr cols (BaseType a)) -> TExpr cols a -> TypedDataFrame cols -> TypedDataFrame cols Source #
Impute missing values in a column using a derived scalar (e.g. the mean).
Schema-preserving: the imputed column keeps its type-level Maybe even though
its runtime values are now fully populated.
summarize :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> DataFrame Source #
Descriptive statistics of the numeric columns. Returns an untyped
DataFrame (the result is a fixed set of statistic rows, not the input schema).
describeColumns :: forall (cols :: [(Symbol, Type)]). TypedDataFrame cols -> DataFrame Source #
Per-column summary (non-null/null counts, unique values, type). Returns an
untyped DataFrame.