{-# LANGUAGE OverloadedStrings #-}

{- | Configuration types for the CSV readers. Re-exported through
"DataFrame.IO.CSV"; shared by the default reader, @dataframe-fastcsv@
and @dataframe-lazy@.
-}
module DataFrame.IO.CSV.Internal.Options (
    HeaderSpec (..),
    TypeSpec (..),
    RaggedRowPolicy (..),
    UnclosedQuotePolicy (..),
    ReadOptions (..),
    defaultReadOptions,
    shouldInferFromSample,
    schemaTypeMap,
    typeInferenceSampleSize,
    resolveSelection,
    InvalidReadOptions (..),
    readOptionsErrors,
    validateReadOptions,
) where

import qualified Data.List as L
import qualified Data.Map.Strict as M
import qualified Data.Text as T

import Control.Exception (Exception, throw, throwIO)
import Control.Monad (unless)
import DataFrame.Errors (DataFrameException (ColumnsNotFoundException))
import DataFrame.Operations.Typing (SafeReadMode (..))
import DataFrame.Schema (SchemaType)

{- | Where a reader's column names come from.

==== __Example__
@
ghci> D.readCsvWithOpts D.defaultReadOptions{D.headerSpec = D.ProvideNames ["a", "b"]} "no_header.csv"

@
-}
data HeaderSpec
    = NoHeader
    | -- | Take names from the first row (the default).
      UseFirstRow
    | -- | Name the leading columns; any beyond the given list are numbered.
      ProvideNames [T.Text]
    deriving (HeaderSpec -> HeaderSpec -> Bool
(HeaderSpec -> HeaderSpec -> Bool)
-> (HeaderSpec -> HeaderSpec -> Bool) -> Eq HeaderSpec
forall a. (a -> a -> Bool) -> (a -> a -> Bool) -> Eq a
$c== :: HeaderSpec -> HeaderSpec -> Bool
== :: HeaderSpec -> HeaderSpec -> Bool
$c/= :: HeaderSpec -> HeaderSpec -> Bool
/= :: HeaderSpec -> HeaderSpec -> Bool
Eq, Int -> HeaderSpec -> ShowS
[HeaderSpec] -> ShowS
HeaderSpec -> String
(Int -> HeaderSpec -> ShowS)
-> (HeaderSpec -> String)
-> ([HeaderSpec] -> ShowS)
-> Show HeaderSpec
forall a.
(Int -> a -> ShowS) -> (a -> String) -> ([a] -> ShowS) -> Show a
$cshowsPrec :: Int -> HeaderSpec -> ShowS
showsPrec :: Int -> HeaderSpec -> ShowS
$cshow :: HeaderSpec -> String
show :: HeaderSpec -> String
$cshowList :: [HeaderSpec] -> ShowS
showList :: [HeaderSpec] -> ShowS
Show)

{- | How a reader decides each column's type.

==== __Example__
@
ghci> D.readCsvWithOpts D.defaultReadOptions{D.typeSpec = D.InferFromSample 500} "wide.csv"

@
-}
data TypeSpec
    = -- | Infer each column's type from the first @n@ rows.
      InferFromSample Int
    | {- | Pin the listed columns to their given types; fall back to the
      nested 'TypeSpec' for the rest.
      -}
      SpecifyTypes [(T.Text, SchemaType)] TypeSpec
    | -- | Every column is read as 'Data.Text.Text', no inference at all.
      NoInference

{- | How the fast reader should treat a row whose field count does not
match the header row.  Only consulted by @dataframe-fastcsv@; the pure
Haskell reader always pads short rows with nulls and drops extras.
-}
data RaggedRowPolicy
    = {- | Fill missing cells with nulls; silently drop extras.
      Matches pandas / polars lenient defaults.
      -}
      PadWithNull
    | {- | Fill missing cells with nulls; silently drop extras.  Alias
      kept for ergonomic naming when the caller only cares about the
      "don't raise, just forget the extras" half of 'PadWithNull'.
      -}
      Truncate
    | {- | Raise a 'DataFrame.IO.CSV.Fast.CsvParseError' on any row whose
      field count differs from the header.  Matches polars' strict
      schema-bound mode.
      -}
      RaiseOnRagged
    deriving (RaggedRowPolicy -> RaggedRowPolicy -> Bool
(RaggedRowPolicy -> RaggedRowPolicy -> Bool)
-> (RaggedRowPolicy -> RaggedRowPolicy -> Bool)
-> Eq RaggedRowPolicy
forall a. (a -> a -> Bool) -> (a -> a -> Bool) -> Eq a
$c== :: RaggedRowPolicy -> RaggedRowPolicy -> Bool
== :: RaggedRowPolicy -> RaggedRowPolicy -> Bool
$c/= :: RaggedRowPolicy -> RaggedRowPolicy -> Bool
/= :: RaggedRowPolicy -> RaggedRowPolicy -> Bool
Eq, Int -> RaggedRowPolicy -> ShowS
[RaggedRowPolicy] -> ShowS
RaggedRowPolicy -> String
(Int -> RaggedRowPolicy -> ShowS)
-> (RaggedRowPolicy -> String)
-> ([RaggedRowPolicy] -> ShowS)
-> Show RaggedRowPolicy
forall a.
(Int -> a -> ShowS) -> (a -> String) -> ([a] -> ShowS) -> Show a
$cshowsPrec :: Int -> RaggedRowPolicy -> ShowS
showsPrec :: Int -> RaggedRowPolicy -> ShowS
$cshow :: RaggedRowPolicy -> String
show :: RaggedRowPolicy -> String
$cshowList :: [RaggedRowPolicy] -> ShowS
showList :: [RaggedRowPolicy] -> ShowS
Show)

-- | How the fast reader should treat an unclosed quoted field at EOF.
data UnclosedQuotePolicy
    = -- | Raise a 'DataFrame.IO.CSV.Fast.CsvParseError'.  Default.
      RaiseOnUnclosedQuote
    | {- | Return whatever rows were parsed before the stray quote; the
      remainder is silently dropped.
      -}
      BestEffort
    deriving (UnclosedQuotePolicy -> UnclosedQuotePolicy -> Bool
(UnclosedQuotePolicy -> UnclosedQuotePolicy -> Bool)
-> (UnclosedQuotePolicy -> UnclosedQuotePolicy -> Bool)
-> Eq UnclosedQuotePolicy
forall a. (a -> a -> Bool) -> (a -> a -> Bool) -> Eq a
$c== :: UnclosedQuotePolicy -> UnclosedQuotePolicy -> Bool
== :: UnclosedQuotePolicy -> UnclosedQuotePolicy -> Bool
$c/= :: UnclosedQuotePolicy -> UnclosedQuotePolicy -> Bool
/= :: UnclosedQuotePolicy -> UnclosedQuotePolicy -> Bool
Eq, Int -> UnclosedQuotePolicy -> ShowS
[UnclosedQuotePolicy] -> ShowS
UnclosedQuotePolicy -> String
(Int -> UnclosedQuotePolicy -> ShowS)
-> (UnclosedQuotePolicy -> String)
-> ([UnclosedQuotePolicy] -> ShowS)
-> Show UnclosedQuotePolicy
forall a.
(Int -> a -> ShowS) -> (a -> String) -> ([a] -> ShowS) -> Show a
$cshowsPrec :: Int -> UnclosedQuotePolicy -> ShowS
showsPrec :: Int -> UnclosedQuotePolicy -> ShowS
$cshow :: UnclosedQuotePolicy -> String
show :: UnclosedQuotePolicy -> String
$cshowList :: [UnclosedQuotePolicy] -> ShowS
showList :: [UnclosedQuotePolicy] -> ShowS
Show)

-- | CSV read parameters.
data ReadOptions = ReadOptions
    { ReadOptions -> HeaderSpec
headerSpec :: HeaderSpec
    -- ^ Where to get the headers from. (default: UseFirstRow)
    , ReadOptions -> TypeSpec
typeSpec :: TypeSpec
    -- ^ Whether/how to infer types. (default: InferFromSample 100)
    , ReadOptions -> SafeReadMode
safeRead :: SafeReadMode
    {- ^ Default 'SafeReadMode' for columns without an entry in
    'safeReadOverrides'. (default: 'NoSafeRead')
    -}
    , ReadOptions -> [(Text, SafeReadMode)]
safeReadOverrides :: [(T.Text, SafeReadMode)]
    -- ^ Per-column 'SafeReadMode' overrides; takes precedence over 'safeRead'.
    , ReadOptions -> String
dateFormat :: String
    {- ^ Format of date fields as recognized by the Data.Time.Format module.

    __Examples:__

    @
    > parseTimeM True defaultTimeLocale "%Y/%-m/%-d" "2010/3/04" :: Maybe Day
    Just 2010-03-04
    > parseTimeM True defaultTimeLocale "%d/%-m/%-Y" "04/3/2010" :: Maybe Day
    Just 2010-03-04
    @
    -}
    , ReadOptions -> Char
columnSeparator :: Char
    -- ^ Character that separates column values.
    , ReadOptions -> Maybe Int
numRowsToRead :: Maybe Int
    {- ^ Maximum number of data rows to read ('Nothing' = all).

    __Example:__

    @
    ghci> D.readCsvWithOpts D.defaultReadOptions{D.numRowsToRead = Just 5} "big.csv"
    @
    -}
    , ReadOptions -> Maybe [Text]
readColumns :: Maybe [T.Text]
    {- ^ Columns to read ('Nothing' = all). Unselected columns are never
    decoded, parsed or allocated; the returned frame carries the selected
    columns in the order given here. Naming a column the file does not have
    raises a 'ColumnsNotFoundException'.

    __Example:__

    @
    ghci> D.readCsvWithOpts D.defaultReadOptions{D.readColumns = Just ["id", "name"]} "customers.csv"
    @
    -}
    , ReadOptions -> [Text]
missingIndicators :: [T.Text]
    -- ^ Values that should be read as `Nothing`.
    , ReadOptions -> RaggedRowPolicy
fastCsvOnRaggedRow :: RaggedRowPolicy
    {- ^ @dataframe-fastcsv@: how to treat rows with a non-header field count.
    (default: 'PadWithNull')
    -}
    , ReadOptions -> UnclosedQuotePolicy
fastCsvOnUnclosedQuote :: UnclosedQuotePolicy
    {- ^ @dataframe-fastcsv@: how to treat an unclosed quoted field at EOF.
    (default: 'RaiseOnUnclosedQuote')
    -}
    , ReadOptions -> Bool
fastCsvTrimUnquoted :: Bool
    {- ^ @dataframe-fastcsv@: if 'True', leading/trailing whitespace is
    stripped from unquoted fields after decoding.  RFC 4180 preserves
    this whitespace, and that is the default ('False').
    -}
    }

{- | Whether a 'TypeSpec' still infers types from a sample, once any
'SpecifyTypes' fallback chain is followed to its end.

==== __Example__
>>> shouldInferFromSample (InferFromSample 100)
True
>>> shouldInferFromSample NoInference
False
-}
shouldInferFromSample :: TypeSpec -> Bool
shouldInferFromSample :: TypeSpec -> Bool
shouldInferFromSample (InferFromSample Int
_) = Bool
True
shouldInferFromSample (SpecifyTypes [(Text, SchemaType)]
_ TypeSpec
fallback) = TypeSpec -> Bool
shouldInferFromSample TypeSpec
fallback
shouldInferFromSample TypeSpec
_ = Bool
False

{- | The explicit column\/type pairs a 'TypeSpec' declares, if any.

==== __Example__
>>> :set -XTypeApplications
>>> M.toList (schemaTypeMap (SpecifyTypes [("id", schemaType @Int)] (InferFromSample 100)))
[("id",Int)]
-}
schemaTypeMap :: TypeSpec -> M.Map T.Text SchemaType
schemaTypeMap :: TypeSpec -> Map Text SchemaType
schemaTypeMap (SpecifyTypes [(Text, SchemaType)]
xs TypeSpec
_) = [(Text, SchemaType)] -> Map Text SchemaType
forall k a. Ord k => [(k, a)] -> Map k a
M.fromList [(Text, SchemaType)]
xs
schemaTypeMap TypeSpec
_ = Map Text SchemaType
forall k a. Map k a
M.empty

{- | A 'ReadOptions' value that contradicts itself, with every problem listed.
Thrown by 'validateReadOptions'; see 'readOptionsErrors' for the check
without the exception.

==== __Example__
>>> show (InvalidReadOptions ["readColumns is an empty selection; omit it to read every column"])
"Invalid ReadOptions:\n  - readColumns is an empty selection; omit it to read every column"
-}
newtype InvalidReadOptions = InvalidReadOptions [T.Text]

instance Show InvalidReadOptions where
    show :: InvalidReadOptions -> String
show (InvalidReadOptions [Text]
problems) =
        Text -> String
T.unpack
            (Text -> [Text] -> Text
T.intercalate Text
"\n" (Text
"Invalid ReadOptions:" Text -> [Text] -> [Text]
forall a. a -> [a] -> [a]
: (Text -> Text) -> [Text] -> [Text]
forall a b. (a -> b) -> [a] -> [b]
map (Text
"  - " Text -> Text -> Text
forall a. Semigroup a => a -> a -> a
<>) [Text]
problems))

instance Exception InvalidReadOptions

{- | Everything self-contradictory about @opts@, judged without looking at a
file. A type or safe-read override naming an unread column is redundant, not
contradictory, and is not reported.

==== __Example__
>>> readOptionsErrors defaultReadOptions{readColumns = Just []}
["readColumns is an empty selection; omit it to read every column"]
>>> readOptionsErrors defaultReadOptions
[]
-}
readOptionsErrors :: ReadOptions -> [T.Text]
readOptionsErrors :: ReadOptions -> [Text]
readOptionsErrors ReadOptions
opts =
    [[Text]] -> [Text]
forall (t :: * -> *) a. Foldable t => t [a] -> [a]
concat
        [ [ Text
"readColumns is an empty selection; omit it to read every column"
          | Just [] <- [ReadOptions -> Maybe [Text]
readColumns ReadOptions
opts]
          ]
        , [ Text
"readColumns names "
                Text -> Text -> Text
forall a. Semigroup a => a -> a -> a
<> Text -> [Text] -> Text
T.intercalate Text
", " [Text]
dups
                Text -> Text -> Text
forall a. Semigroup a => a -> a -> a
<> Text
" more than once"
          | Bool -> Bool
not ([Text] -> Bool
forall a. [a] -> Bool
forall (t :: * -> *) a. Foldable t => t a -> Bool
null [Text]
dups)
          ]
        , [ Text
"numRowsToRead is negative: " Text -> Text -> Text
forall a. Semigroup a => a -> a -> a
<> String -> Text
T.pack (Int -> String
forall a. Show a => a -> String
show Int
n)
          | Just Int
n <- [ReadOptions -> Maybe Int
numRowsToRead ReadOptions
opts]
          , Int
n Int -> Int -> Bool
forall a. Ord a => a -> a -> Bool
< Int
0
          ]
        , [ Text
"headerSpec is ProvideNames with no names"
          | ProvideNames [] <- [ReadOptions -> HeaderSpec
headerSpec ReadOptions
opts]
          ]
        , [ Text
"columnSeparator is " Text -> Text -> Text
forall a. Semigroup a => a -> a -> a
<> String -> Text
T.pack (Char -> String
forall a. Show a => a -> String
show Char
sep) Text -> Text -> Text
forall a. Semigroup a => a -> a -> a
<> Text
", which cannot delimit fields"
          | let sep :: Char
sep = ReadOptions -> Char
columnSeparator ReadOptions
opts
          , Char
sep Char -> String -> Bool
forall a. Eq a => a -> [a] -> Bool
forall (t :: * -> *) a. (Foldable t, Eq a) => a -> t a -> Bool
`elem` [Char
'"', Char
'\n', Char
'\r']
          ]
        ]
  where
    dups :: [Text]
dups = [Text] -> ([Text] -> [Text]) -> Maybe [Text] -> [Text]
forall b a. b -> (a -> b) -> Maybe a -> b
maybe [] (\[Text]
cs -> [Text] -> [Text]
forall a. Eq a => [a] -> [a]
L.nub ([Text]
cs [Text] -> [Text] -> [Text]
forall a. Eq a => [a] -> [a] -> [a]
L.\\ [Text] -> [Text]
forall a. Eq a => [a] -> [a]
L.nub [Text]
cs)) (ReadOptions -> Maybe [Text]
readColumns ReadOptions
opts)

{- | Throw 'InvalidReadOptions' if @opts@ contradicts itself. Both readers
call this before opening the file.

==== __Example__
@
ghci> D.validateReadOptions D.defaultReadOptions{D.readColumns = Just []}
*** Exception: Invalid ReadOptions:
  - readColumns is an empty selection; omit it to read every column
@
-}
validateReadOptions :: ReadOptions -> IO ()
validateReadOptions :: ReadOptions -> IO ()
validateReadOptions ReadOptions
opts =
    Bool -> IO () -> IO ()
forall (f :: * -> *). Applicative f => Bool -> f () -> f ()
unless ([Text] -> Bool
forall a. [a] -> Bool
forall (t :: * -> *) a. Foldable t => t a -> Bool
null [Text]
problems) (InvalidReadOptions -> IO ()
forall e a. Exception e => e -> IO a
throwIO ([Text] -> InvalidReadOptions
InvalidReadOptions [Text]
problems))
  where
    problems :: [Text]
problems = ReadOptions -> [Text]
readOptionsErrors ReadOptions
opts

{- | Resolve 'readColumns' against a file's header names: the columns to
build, each paired with its field index within a row. 'Nothing' keeps every
column in file order; a selection keeps the order it was written in.

==== __Example__
>>> resolveSelection defaultReadOptions{readColumns = Just ["b", "a"]} ["a", "b", "c"]
[("b",1),("a",0)]
>>> resolveSelection defaultReadOptions ["a", "b"]
[("a",0),("b",1)]
-}
resolveSelection :: ReadOptions -> [T.Text] -> [(T.Text, Int)]
resolveSelection :: ReadOptions -> [Text] -> [(Text, Int)]
resolveSelection ReadOptions
opts [Text]
names = case ReadOptions -> Maybe [Text]
readColumns ReadOptions
opts of
    Maybe [Text]
Nothing -> [(Text, Int)]
indexed
    Just [Text]
requested -> case (Text -> Bool) -> [Text] -> [Text]
forall a. (a -> Bool) -> [a] -> [a]
filter (Text -> Map Text Int -> Bool
forall k a. Ord k => k -> Map k a -> Bool
`M.notMember` Map Text Int
fieldIndex) [Text]
requested of
        [] -> [(Text
n, Map Text Int
fieldIndex Map Text Int -> Text -> Int
forall k a. Ord k => Map k a -> k -> a
M.! Text
n) | Text
n <- [Text]
requested]
        [Text]
missing -> DataFrameException -> [(Text, Int)]
forall a e. Exception e => e -> a
throw ([Text] -> Text -> [Text] -> DataFrameException
ColumnsNotFoundException [Text]
missing Text
"readCsvWithOpts" [Text]
names)
  where
    indexed :: [(Text, Int)]
indexed = [Text] -> [Int] -> [(Text, Int)]
forall a b. [a] -> [b] -> [(a, b)]
zip [Text]
names [Int
0 ..]
    fieldIndex :: Map Text Int
fieldIndex = [(Text, Int)] -> Map Text Int
forall k a. Ord k => [(k, a)] -> Map k a
M.fromList [(Text, Int)]
indexed

{- | The sample size a 'TypeSpec' infers from, once any 'SpecifyTypes'
fallback chain is followed to its end. @0@ when the chain ends in
'NoInference'.

==== __Example__
>>> typeInferenceSampleSize (InferFromSample 100)
100
>>> typeInferenceSampleSize NoInference
0
-}
typeInferenceSampleSize :: TypeSpec -> Int
typeInferenceSampleSize :: TypeSpec -> Int
typeInferenceSampleSize (InferFromSample Int
n) = Int
n
typeInferenceSampleSize (SpecifyTypes [(Text, SchemaType)]
_ TypeSpec
fallback) = TypeSpec -> Int
typeInferenceSampleSize TypeSpec
fallback
typeInferenceSampleSize TypeSpec
_ = Int
0

{- | The default 'ReadOptions': infer types from a 100-row sample, treat the
first row as a header, read every row and every column.

==== __Example__
@
ghci> D.readCsvWithOpts D.defaultReadOptions{D.columnSeparator = ';'} "data.csv"
@
-}
defaultReadOptions :: ReadOptions
defaultReadOptions :: ReadOptions
defaultReadOptions =
    ReadOptions
        { headerSpec :: HeaderSpec
headerSpec = HeaderSpec
UseFirstRow
        , typeSpec :: TypeSpec
typeSpec = Int -> TypeSpec
InferFromSample Int
100
        , safeRead :: SafeReadMode
safeRead = SafeReadMode
NoSafeRead
        , safeReadOverrides :: [(Text, SafeReadMode)]
safeReadOverrides = []
        , dateFormat :: String
dateFormat = String
"%Y-%m-%d"
        , columnSeparator :: Char
columnSeparator = Char
','
        , numRowsToRead :: Maybe Int
numRowsToRead = Maybe Int
forall a. Maybe a
Nothing
        , readColumns :: Maybe [Text]
readColumns = Maybe [Text]
forall a. Maybe a
Nothing
        , missingIndicators :: [Text]
missingIndicators =
            [Text
"Nothing", Text
"NULL", Text
"", Text
" ", Text
"nan", Text
"null", Text
"N/A", Text
"NaN", Text
"NAN", Text
"NA"]
        , fastCsvOnRaggedRow :: RaggedRowPolicy
fastCsvOnRaggedRow = RaggedRowPolicy
PadWithNull
        , fastCsvOnUnclosedQuote :: UnclosedQuotePolicy
fastCsvOnUnclosedQuote = UnclosedQuotePolicy
RaiseOnUnclosedQuote
        , fastCsvTrimUnquoted :: Bool
fastCsvTrimUnquoted = Bool
False
        }