dataframe-operations-2.4.0.0: Column operations, expression DSL, and statistics for the dataframe ecosystem.
Safe HaskellNone
LanguageHaskell2010

DataFrame.Operations.Core

Synopsis

Dimensions

dimensions :: DataFrame -> (Int, Int) Source #

O(1) Get DataFrame dimensions i.e. (rows, columns)

Example

Expand
>>> :set -XOverloadedStrings
>>> import qualified DataFrame as D
>>> df = D.fromNamedColumns [("a", D.fromList [1..100]), ("b", D.fromList [1..100]), ("c", D.fromList [1..100])]
>>> D.dimensions df

(100, 3)

nRows :: DataFrame -> Int Source #

O(1) Get number of rows in a dataframe.

Example

Expand
>>> :set -XOverloadedStrings
>>> import qualified DataFrame as D
>>> df = D.fromNamedColumns [("a", D.fromList [1..100]), ("b", D.fromList [1..100]), ("c", D.fromList [1..100])]
>>> D.nRows df
100

nColumns :: DataFrame -> Int Source #

O(1) Get number of columns in a dataframe.

Example

Expand
>>> :set -XOverloadedStrings
>>> import qualified DataFrame as D
>>> df = D.fromNamedColumns [("a", D.fromList [1..100]), ("b", D.fromList [1..100]), ("c", D.fromList [1..100])]
>>> D.nColumns df
3

Construction

fromUnnamedColumns :: [Column] -> DataFrame Source #

Creates a dataframe from a list of tuples with name and column.

Example

Expand
>>> df = D.fromNamedColumns [("numbers", D.fromList [1..10]), ("others", D.fromList [11..20])]
>>> df
-----------------
 numbers | others
---------|-------
   Int   |  Int
---------|-------
 1       | 11
 2       | 12
 3       | 13
 4       | 14
 5       | 15
 6       | 16
 7       | 17
 8       | 18
 9       | 19
 10      | 20

Create a dataframe from a list of columns. The column names are "0", "1"... etc. Useful for quick exploration but you should probably always rename the columns after or drop the ones you don't want.

Example

Expand
>>> df = D.fromUnnamedColumns [D.fromList [1..10], D.fromList [11..20]]
>>> df
-----------------
  0  |  1
-----|----
 Int | Int
-----|----
 1   | 11
 2   | 12
 3   | 13
 4   | 14
 5   | 15
 6   | 16
 7   | 17
 8   | 18
 9   | 19
 10  | 20

fromRows :: [Text] -> [[Any]] -> DataFrame Source #

Create a dataframe from a list of column names and rows.

Example

Expand
>>> df = D.fromRows [A, B] [[D.toAny 1, D.toAny 11], [D.toAny 2, D.toAny 12], [D.toAny 3, D.toAny 13]]

>>> df

----------
  A  |  B
-----|----
 Int | Int
-----|----
 1   | 11
 2   | 12
 3   | 13

Insertion

insert Source #

Arguments

:: (Columnable a, Foldable t) 
=> Text

Column Name

-> t a

Sequence to add to dataframe

-> DataFrame

DataFrame to add column to

-> DataFrame 

Adds a foldable collection as a named column. Size mismatches are reconciled by making the shorter side nullable (`Maybe a`) and padding with Nothing. Do not pass infinite collections: they are fully forced.

Example

Expand
>>> :set -XOverloadedStrings
>>> import qualified DataFrame as D
>>> D.insert "numbers" [(1 :: Int)..10] D.empty

--------
 numbers
--------
   Int
--------
 1
 2
 3
 4
 5
 6
 7
 8
 9
 10

insertVector Source #

Arguments

:: Columnable a 
=> Text

Column Name

-> Vector a

Vector to add to column

-> DataFrame

DataFrame to add column to

-> DataFrame 

O(k) Get column names of the DataFrame in order of insertion.

Example

Expand
>>> :set -XOverloadedStrings
>>> import qualified DataFrame as D
>>> df = D.fromNamedColumns [("a", D.fromList [1..100]), ("b", D.fromList [1..100]), ("c", D.fromList [1..100])]
>>> D.columnNames df

["a", "b", "c"]

Adds a vector as a named column. Size mismatches are reconciled by making the shorter side nullable (`Maybe a`) and padding with Nothing.

Example

Expand
>>> :set -XOverloadedStrings
>>> import qualified DataFrame as D
>>> import qualified Data.Vector as V
>>> D.insertVector "numbers" (V.fromList [(1 :: Int)..10]) D.empty

--------
 numbers
--------
   Int
--------
 1
 2
 3
 4
 5
 6
 7
 8
 9
 10

insertUnboxedVector Source #

Arguments

:: (Columnable a, Unbox a) 
=> Text

Column Name

-> Vector a

Unboxed vector to add to column

-> DataFrame

DataFrame to add the column to

-> DataFrame 

O(n) Like insertVector but takes an already-unboxed vector, skipping the boxed-to-unboxed conversion insertVector would do for numbers.

insertWithDefault Source #

Arguments

:: (Columnable a, Foldable t) 
=> a

Default Value

-> Text

Column name

-> t a

Data to add to column

-> DataFrame

DataFrame to add the column to

-> DataFrame 

Adds a list to the dataframe and pads it with a default value if it has less elements than the number of rows.

Example

Expand
>>> :set -XOverloadedStrings
>>> import qualified DataFrame as D
>>> df = D.fromNamedColumns [("x", D.fromList [(1 :: Int)..10])]
>>> D.insertWithDefault 0 "numbers" [(1 :: Int),2,3] df

-------------
 x  | numbers
----|--------
Int |   Int
----|--------
1   | 1
2   | 2
3   | 3
4   | 0
5   | 0
6   | 0
7   | 0
8   | 0
9   | 0
10  | 0

insertVectorWithDefault Source #

Arguments

:: Columnable a 
=> a

Default Value

-> Text

Column name

-> Vector a

Data to add to column

-> DataFrame

DataFrame to add the column to

-> DataFrame 

Adds a vector to the dataframe and pads it with a default value if it has less elements than the number of rows.

Example

Expand
>>> :set -XOverloadedStrings
>>> import qualified Data.Vector as V
>>> import qualified DataFrame as D
>>> df = D.fromNamedColumns [("x", D.fromList [(1 :: Int)..10])]
>>> D.insertVectorWithDefault 0 "numbers" (V.fromList [(1 :: Int),2,3]) df

-------------
 x  | numbers
----|--------
Int |   Int
----|--------
1   | 1
2   | 2
3   | 3
4   | 0
5   | 0
6   | 0
7   | 0
8   | 0
9   | 0
10  | 0

Column management

cloneColumn :: Text -> Text -> DataFrame -> DataFrame Source #

O(n) Add a column to the dataframe.

Example

Expand
>>> :set -XOverloadedStrings
>>> import qualified DataFrame as D
>>> D.insertColumn "numbers" (D.fromList [(1 :: Int)..10]) D.empty

--------
 numbers
--------
   Int
--------
 1
 2
 3
 4
 5
 6
 7
 8
 9
 10

O(n) Clones a column and places it under a new name in the dataframe.

Example

Expand
>>> :set -XOverloadedStrings
>>> import qualified Data.Vector as V
>>> df = insertVector "numbers" (V.fromList [1..10]) D.empty
>>> D.cloneColumn "numbers" "others" df

-----------------
 numbers | others
---------|-------
   Int   |  Int
---------|-------
 1       | 1
 2       | 2
 3       | 3
 4       | 4
 5       | 5
 6       | 6
 7       | 7
 8       | 8
 9       | 9
 10      | 10

rename :: Text -> Text -> DataFrame -> DataFrame Source #

O(n) Renames a single column.

Example

Expand
>>> :set -XOverloadedStrings
>>> import qualified DataFrame as D
>>> import qualified Data.Vector as V
>>> df = insertVector "numbers" (V.fromList [1..10]) D.empty
>>> D.rename "numbers" "others" df

-------
 others
-------
  Int
-------
 1
 2
 3
 4
 5
 6
 7
 8
 9
 10

renameMany :: [(Text, Text)] -> DataFrame -> DataFrame Source #

O(n) Renames many columns.

Example

Expand
>>> :set -XOverloadedStrings
>>> import qualified DataFrame as D
>>> import qualified Data.Vector as V
>>> df = D.insertVector "others" (V.fromList [11..20]) (D.insertVector "numbers" (V.fromList [1..10]) D.empty)
>>> df

-----------------
 numbers | others
---------|-------
   Int   |  Int
---------|-------
 1       | 11
 2       | 12
 3       | 13
 4       | 14
 5       | 15
 6       | 16
 7       | 17
 8       | 18
 9       | 19
 10      | 20

>>> D.renameMany [("numbers", "first_10"), ("others", "next_10")] df

-------------------
 first_10 | next_10
----------|--------
   Int    |   Int
----------|--------
 1        | 11
 2        | 12
 3        | 13
 4        | 14
 5        | 15
 6        | 16
 7        | 17
 8        | 18
 9        | 19
 10       | 20

Inspection

describeColumns :: DataFrame -> DataFrame Source #

O(n * k ^ 2) Returns the number of non-null columns in the dataframe and the type associated with each column.

Example

Expand
>>> import qualified Data.Vector as V
>>> df = D.insertVector "others" (V.fromList [11..20]) (D.insertVector "numbers" (V.fromList [1..10]) D.empty)
>>> D.describeColumns df

--------------------------------------------------------
 Column Name | # Non-null Values | # Null Values | Type
-------------|-------------------|---------------|-----
    Text     |        Int        |      Int      | Text
-------------|-------------------|---------------|-----
 others      | 10                | 0             | Int
 numbers     | 10                | 0             | Int

valueCounts :: (Ord a, Columnable a) => Expr a -> DataFrame -> [(a, Int)] Source #

O (k * n) Counts the occurences of each value in a given column.

Example

Expand
>>> df = D.fromUnnamedColumns [D.fromList [1..10], D.fromList [11..20]]

>>> D.valueCounts @Int "0" df

[(1,1),(2,1),(3,1),(4,1),(5,1),(6,1),(7,1),(8,1),(9,1),(10,1)]

valueProportions :: (Ord a, Columnable a) => Expr a -> DataFrame -> [(a, Double)] Source #

O (k * n) Shows the proportions of each value in a given column.

Example

Expand
>>> df = D.fromUnnamedColumns [D.fromList [1..10], D.fromList [11..20]]

>>> D.valueCounts @Int "0" df

[(1,0.1),(2,0.1),(3,0.1),(4,0.1),(5,0.1),(6,0.1),(7,0.1),(8,0.1),(9,0.1),(10,0.1)]

showDerivedExpressions :: DataFrame -> [NamedExpr] Source #

Returns the provenance of all columns in the DataFrame as a list of (name, expression) pairs. Derived columns show their expression; raw columns show an identity col @type name expression.

Folds & matrix/vector conversions

fold :: (a -> DataFrame -> DataFrame) -> [a] -> DataFrame -> DataFrame Source #

A left fold for dataframes that takes the dataframe as the last object. This makes it easier to chain operations.

Example

Expand
>>> df = D.fromNamedColumns [("x", D.fromList [1..100]), ("y", D.fromList [11..110])]
>>> D.fold D.dropLast [1..5] df

---------
 x  |  y
----|----
Int | Int
----|----
1   | 11
2   | 12
3   | 13
4   | 14
5   | 15
6   | 16
7   | 17
8   | 18
9   | 19
10  | 20
11  | 21
12  | 22
13  | 23
14  | 24
15  | 25
16  | 26
17  | 27
18  | 28
19  | 29
20  | 30

Showing 20 rows out of 85

toFloatMatrix :: DataFrame -> Either DataFrameException (Vector (Vector Float)) Source #

The dataframe as a row-major matrix of floats, for handing data to ML systems. Left if any column cannot be converted to floats.

toDoubleMatrix :: DataFrame -> Either DataFrameException (Vector (Vector Double)) Source #

The dataframe as a row-major matrix of doubles, for handing data to ML systems. Left if any column cannot be converted to doubles.

toIntMatrix :: DataFrame -> Either DataFrameException (Vector (Vector Int)) Source #

The dataframe as a row-major matrix of ints, for handing data to ML systems. Left if any column cannot be converted to ints.

columnAsVector :: Columnable a => Expr a -> DataFrame -> Either DataFrameException (Vector a) Source #

Get a specific column as a vector.

You must specify the type via type applications.

Examples

Expand
>>> columnAsVector (F.col @Int "age") df
Right [25, 30, 35, ...]
>>> columnAsVector (F.col @Text "name") df
Right ["Alice", "Bob", "Charlie", ...]

columnAsList :: Columnable a => Expr a -> DataFrame -> [a] Source #

Get a specific column as a list.

You must specify the type via type applications.

Examples

Expand
>>> columnAsList @Int "age" df
[25, 30, 35, ...]
>>> columnAsList @Text "name" df
["Alice", "Bob", "Charlie", ...]

Throws

Expand
  • error - if the column type doesn't match the requested type

columnAsIntVector :: (Columnable a, Num a) => Expr a -> DataFrame -> Either DataFrameException (Vector Int) Source #

A column as an unboxed vector of Int values. Left if the column cannot be converted to ints (non-numeric or out of range).

columnAsDoubleVector :: (Columnable a, Num a) => Expr a -> DataFrame -> Either DataFrameException (Vector Double) Source #

A column as an unboxed vector of Double values. Left if the column cannot be converted to doubles (e.g. non-numeric data).

columnAsFloatVector :: (Columnable a, Num a) => Expr a -> DataFrame -> Either DataFrameException (Vector Float) Source #

A column as an unboxed vector of Float values. Left if the column cannot be converted to floats (e.g. non-numeric data).